{"version":"v4-configuration-source-5","legacyPooledVersion":"v4-2026-09-07-overall-3","capabilityVersion":"v4-2026-09-07-v4-core-4","defaultMode":"overall","overall":{"version":"configuration-source-5","smoothing":0.2,"maximumLabWeightShare":0.25,"minimumIndependentSources":1,"minimumIndependentFamilies":2,"referencePanel":["gpt-5.6-sol--effort-high","claude-opus-4-8--effort-high","gemini-3.1-pro--effort-high"],"minimumFamilies":3,"minimumOpponents":3,"capabilityReferencePanel":["gpt-5.6-sol--effort-high","claude-opus-4-7--effort-max","gemini-3.5-flash--effort-high"]},"registry":{"version":"2026-09-07-v2","referencePanel":["gpt-6-astra","gpt-5.6-sol","gpt-5.6-terra","gpt-5.5","claude-fable-5-1","claude-fable-5","claude-opus-5","claude-opus-4-8","claude-sonnet-4-6","gemini-3.8-flash","gemini-3.1-pro","gemini-2.5-pro","muse-spark-1.3","muse-spark-1.1","deepseek-v4-pro","deepseek-v4-flash","qwen-3.8-max","qwen3.5-397b-a17b","grok-4.6","grok-4.5","kimi-k3","kimi-k2.6","minimax-m3","minimax-m2.7","glm-5.3","glm-5.1","mistral-medium-3.5","magistral-medium-2509","nemotron-3-ultra","amazon-nova-2-pro-preview","amazon-nova-premier","command-a-plus-05-2026","command-a-reasoning-08-2025","mai-thinking-1"],"families":{"$OneMillion-Bench (expert score)":{"familyId":"onemillion-bench-expert-score","label":"Onemillion Bench Expert Score","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"AA Intelligence Index":{"familyId":"aa-intelligence-index","label":"Aa Intelligence Index","capability":"agentic","included":false,"reason":"Composite index or publisher-specific internal evaluation lacks a stable general-capability comparison unit."},"AA-Briefcase":{"familyId":"aa-briefcase","label":"AA-Briefcase","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"AA-Briefcase (Elo)":{"familyId":"aa-briefcase","label":"AA-Briefcase","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"AA-LCR":{"familyId":"aa-lcr","label":"Aa Lcr","capability":"long_context","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"AIME":{"familyId":"aime","label":"Aime","capability":"hard_reasoning","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"APEX-Agents":{"familyId":"apex-agents","label":"Apex Agents","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"APEX-SWE":{"familyId":"apex-swe","label":"Apex Swe","capability":"coding","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"ARC-AGI":{"familyId":"arc-agi","label":"ARC-AGI","capability":"hard_reasoning","included":true,"reason":"ARC generations share one family weight; task versions remain separate comparison units and cannot be paired across generations."},"ARC-AGI-1":{"familyId":"arc-agi","label":"ARC-AGI","capability":"hard_reasoning","included":true,"reason":"ARC generations share one family weight; task versions remain separate comparison units and cannot be paired across generations."},"ARC-AGI-2":{"familyId":"arc-agi","label":"ARC-AGI","capability":"hard_reasoning","included":true,"reason":"ARC generations share one family weight; task versions remain separate comparison units and cannot be paired across generations."},"ARC-AGI-3":{"familyId":"arc-agi","label":"ARC-AGI","capability":"hard_reasoning","included":true,"reason":"ARC generations share one family weight; task versions remain separate comparison units and cannot be paired across generations."},"AdvancedIF rubric-level":{"familyId":"advancedif-rubric-level","label":"Advancedif Rubric Level","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"Agentic IF Index":{"familyId":"agentic-if-index","label":"Agentic If Index","capability":"agentic","included":false,"reason":"Composite index or publisher-specific internal evaluation lacks a stable general-capability comparison unit."},"Agents' Last Exam":{"familyId":"agents-last-exam","label":"Agents’ Last Exam","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"Agents' Last Exam (ALE-CLI)":{"familyId":"agents-last-exam","label":"Agents’ Last Exam","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"Agents' Last Exam (Pass / Score)":{"familyId":"agents-last-exam","label":"Agents’ Last Exam","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"AllenAI IFBench":{"familyId":"ifbench","label":"IFBench","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"AndroidBench":{"familyId":"androidbench","label":"Androidbench","capability":"coding","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"Apex-Shortlist (no tools)":{"familyId":"apex-shortlist","label":"Apex Shortlist","capability":"hard_reasoning","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"Apex-Shortlist (with tools)":{"familyId":"apex-shortlist","label":"Apex Shortlist","capability":"hard_reasoning","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"Artificial Analysis Coding Agent Index v1.1":{"familyId":"artificial-analysis-coding-agent-index-v1-1","label":"Artificial Analysis Coding Agent Index V1 1","capability":"coding","included":false,"reason":"Composite index or publisher-specific internal evaluation lacks a stable general-capability comparison unit."},"Artificial Analysis Coding Agent Index v1.4":{"familyId":"artificial-analysis-coding-agent-index-v1-4","label":"Artificial Analysis Coding Agent Index V1 4","capability":"coding","included":false,"reason":"Composite index or publisher-specific internal evaluation lacks a stable general-capability comparison unit."},"Artificial Analysis Intelligence Index v4.1":{"familyId":"artificial-analysis-intelligence-index-v4-1","label":"Artificial Analysis Intelligence Index V4 1","capability":"agentic","included":false,"reason":"Composite index or publisher-specific internal evaluation lacks a stable general-capability comparison unit."},"Artificial Analysis Intelligence Index v4.1.1":{"familyId":"artificial-analysis-intelligence-index-v4-1-1","label":"Artificial Analysis Intelligence Index V4 1 1","capability":"agentic","included":false,"reason":"Composite index or publisher-specific internal evaluation lacks a stable general-capability comparison unit."},"Automation-Bench (Pass@1)":{"familyId":"automationbench","label":"AutomationBench","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"AutomationBench":{"familyId":"automationbench","label":"AutomationBench","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"AutomationBench (Public)":{"familyId":"automationbench","label":"AutomationBench","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"AutomationBench (v1.0.6)":{"familyId":"automationbench","label":"AutomationBench","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"BFCL v3":{"familyId":"bfcl-v3","label":"Bfcl V3","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"BFCL v4":{"familyId":"bfcl-v4","label":"Bfcl V4","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"BabyVision w/ python":{"familyId":"babyvision","label":"BabyVision","capability":"multimodal","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"BabyVision with tools":{"familyId":"babyvision","label":"BabyVision","capability":"multimodal","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"BankerToolBench":{"familyId":"bankertoolbench","label":"Bankertoolbench","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"BenchCAD":{"familyId":"benchcad","label":"BenchCAD","capability":"multimodal","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"BenchCAD (python tool)":{"familyId":"benchcad","label":"BenchCAD","capability":"multimodal","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"BeyondAIME avg@16":{"familyId":"beyondaime-avg-16","label":"Beyondaime Avg 16","capability":"hard_reasoning","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"Big Finance Bench":{"familyId":"big-finance-bench","label":"Big Finance Bench","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"BioMysteryBench":{"familyId":"biomysterybench","label":"Biomysterybench","capability":"knowledge","included":false,"reason":"Narrow specialist biological, medical or security evaluation; preserved as raw evidence outside general capability scoring."},"BrowseComp":{"familyId":"browsecomp","label":"Browsecomp","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"CL-bench":{"familyId":"cl-bench","label":"Cl Bench","capability":"long_context","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"Capture-the-Flag Challenges":{"familyId":"capture-the-flag-challenges","label":"Capture The Flag Challenges","capability":"coding","included":false,"reason":"Narrow specialist biological, medical or security evaluation; preserved as raw evidence outside general capability scoring."},"CharXiv (RQ)":{"familyId":"charxiv","label":"CharXiv","capability":"multimodal","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"CharXiv Reasoning":{"familyId":"charxiv","label":"CharXiv","capability":"multimodal","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"CharXiv Reasoning with tools":{"familyId":"charxiv","label":"CharXiv","capability":"multimodal","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"CharXiv descriptive":{"familyId":"charxiv","label":"CharXiv","capability":"multimodal","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"Chartography":{"familyId":"chartography","label":"Chartography","capability":"multimodal","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"Claw-Eval":{"familyId":"claw-eval","label":"Claw Eval","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"CoWorkBench":{"familyId":"coworkbench","label":"Coworkbench","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"Collie":{"familyId":"collie","label":"Collie","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"Coronavirus-ACE2 Cell-Entry Screen":{"familyId":"coronavirus-ace2-cell-entry-screen","label":"Coronavirus Ace2 Cell Entry Screen","capability":"knowledge","included":false,"reason":"Narrow specialist biological, medical or security evaluation; preserved as raw evidence outside general capability scoring."},"CorpFin v2":{"familyId":"corpfin-v2","label":"Corpfin V2","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"CorpusQA":{"familyId":"corpusqa","label":"Corpusqa","capability":"long_context","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"CritPt":{"familyId":"critpt","label":"Critpt","capability":"hard_reasoning","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"CritPt (no tools)":{"familyId":"critpt","label":"Critpt","capability":"hard_reasoning","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"CursorBench":{"familyId":"cursorbench","label":"Cursorbench","capability":"coding","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"CursorBench v3.2":{"familyId":"cursorbench","label":"Cursorbench","capability":"coding","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"CyberGym":{"familyId":"cybergym","label":"Cybergym","capability":"coding","included":false,"reason":"Narrow specialist biological, medical or security evaluation; preserved as raw evidence outside general capability scoring."},"Cybergym":{"familyId":"cybergym","label":"Cybergym","capability":"coding","included":false,"reason":"Narrow specialist biological, medical or security evaluation; preserved as raw evidence outside general capability scoring."},"DRACO":{"familyId":"draco","label":"Draco","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"DSBench-FullStack †":{"familyId":"dsbench-fullstack","label":"Dsbench Fullstack","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"DSBench-Hard †":{"familyId":"dsbench-hard","label":"Dsbench Hard","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"DeepSWE":{"familyId":"deepswe","label":"DeepSWE","capability":"coding","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"DeepSWE (v1.1)":{"familyId":"deepswe","label":"DeepSWE","capability":"coding","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"DeepSWE 1.1":{"familyId":"deepswe","label":"DeepSWE","capability":"coding","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"DeepSWE v1.1":{"familyId":"deepswe","label":"DeepSWE","capability":"coding","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"DeepSearchQA":{"familyId":"deepsearchqa","label":"DeepSearchQA","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"DeepSearchQA (F1)":{"familyId":"deepsearchqa","label":"DeepSearchQA","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"ExploitBench":{"familyId":"exploitbench","label":"Exploitbench","capability":"coding","included":false,"reason":"Narrow specialist biological, medical or security evaluation; preserved as raw evidence outside general capability scoring."},"ExploitBench (June-Aug 2026)":{"familyId":"exploitbench-june-aug-2026","label":"Exploitbench June Aug 2026","capability":"coding","included":false,"reason":"Narrow specialist biological, medical or security evaluation; preserved as raw evidence outside general capability scoring."},"ExploitGym":{"familyId":"exploitgym","label":"Exploitgym","capability":"coding","included":false,"reason":"Narrow specialist biological, medical or security evaluation; preserved as raw evidence outside general capability scoring."},"ExploitGym (2h / 6h)":{"familyId":"exploitgym-2h-6h","label":"Exploitgym 2H 6H","capability":"coding","included":false,"reason":"Narrow specialist biological, medical or security evaluation; preserved as raw evidence outside general capability scoring."},"Finance Agent v2":{"familyId":"finance-agent","label":"Finance Agent","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"FrontierCode":{"familyId":"frontiercode","label":"FrontierCode","capability":"coding","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"FrontierCode 1.1 Extended (score)":{"familyId":"frontiercode","label":"FrontierCode","capability":"coding","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"FrontierCode 1.1 Main (score)":{"familyId":"frontiercode","label":"FrontierCode","capability":"coding","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"FrontierCode v1.1 Extended":{"familyId":"frontiercode","label":"FrontierCode","capability":"coding","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"FrontierMath Tier 1-3 (v2)":{"familyId":"frontiermath","label":"FrontierMath","capability":"hard_reasoning","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"FrontierMath Tier 4 (v2)":{"familyId":"frontiermath","label":"FrontierMath","capability":"hard_reasoning","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"FrontierSWE":{"familyId":"frontierswe","label":"FrontierSWE","capability":"coding","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"GDP.PDF":{"familyId":"gdp-pdf","label":"Gdp Pdf","capability":"multimodal","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"GDPVal":{"familyId":"gdpval","label":"GDPval","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"GDPval rubrics":{"familyId":"gdpval","label":"GDPval","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"GDPval-AA":{"familyId":"gdpval","label":"GDPval","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"GDPval-AA v2":{"familyId":"gdpval","label":"GDPval","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"GDPval-AA v2 (Elo)":{"familyId":"gdpval","label":"GDPval","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"GDPval-AA v2 Elo":{"familyId":"gdpval","label":"GDPval","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"GMMLU":{"familyId":"gmmlu","label":"Gmmlu","capability":"knowledge","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"GPQA (no tools)":{"familyId":"gpqa","label":"GPQA","capability":"hard_reasoning","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"GPQA Diamond":{"familyId":"gpqa","label":"GPQA","capability":"hard_reasoning","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"GeneBench Pro":{"familyId":"genebench-pro","label":"Genebench Pro","capability":"knowledge","included":false,"reason":"Narrow specialist biological, medical or security evaluation; preserved as raw evidence outside general capability scoring."},"GraphWalks <=128k":{"familyId":"graphwalks","label":"Graphwalks","capability":"long_context","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"GraphWalks BFS 1mil f1":{"familyId":"graphwalks","label":"Graphwalks","capability":"long_context","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"GraphWalks BFS 256k f1":{"familyId":"graphwalks","label":"Graphwalks","capability":"long_context","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"HMMT February 2026":{"familyId":"hmmt-february-2026","label":"Hmmt February 2026","capability":"hard_reasoning","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"Harvey LAB (Vals)":{"familyId":"harvey-lab","label":"Harvey Lab","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"Harvey Lab-AA":{"familyId":"harvey-lab","label":"Harvey Lab","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"Harvey Legal Agent Benchmark":{"familyId":"harvey-lab","label":"Harvey Lab","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"HealthBench":{"familyId":"healthbench","label":"Healthbench","capability":"knowledge","included":false,"reason":"Narrow specialist biological, medical or security evaluation; preserved as raw evidence outside general capability scoring."},"HealthBench Consensus":{"familyId":"healthbench","label":"Healthbench","capability":"knowledge","included":false,"reason":"Narrow specialist biological, medical or security evaluation; preserved as raw evidence outside general capability scoring."},"HealthBench Hard":{"familyId":"healthbench","label":"Healthbench","capability":"knowledge","included":false,"reason":"Narrow specialist biological, medical or security evaluation; preserved as raw evidence outside general capability scoring."},"HealthBench Professional":{"familyId":"healthbench","label":"Healthbench","capability":"knowledge","included":false,"reason":"Narrow specialist biological, medical or security evaluation; preserved as raw evidence outside general capability scoring."},"HealthBench Professional (length-adjusted)":{"familyId":"healthbench","label":"Healthbench","capability":"knowledge","included":false,"reason":"Narrow specialist biological, medical or security evaluation; preserved as raw evidence outside general capability scoring."},"Humanity's Last Exam":{"familyId":"hle","label":"Humanity’s Last Exam","capability":"hard_reasoning","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"IF Bench":{"familyId":"ifbench","label":"IFBench","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"IFBench":{"familyId":"ifbench","label":"IFBench","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"IFBench (prompt loose)":{"familyId":"ifbench","label":"IFBench","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"IFBench prompt loose":{"familyId":"ifbench","label":"IFBench","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"IMO 2025":{"familyId":"imo-2025","label":"Imo 2025","capability":"hard_reasoning","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"IMOAnswerBench (no tools)":{"familyId":"imo-answerbench","label":"Imo Answerbench","capability":"hard_reasoning","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"IMOAnswerBench (with tools)":{"familyId":"imo-answerbench","label":"Imo Answerbench","capability":"hard_reasoning","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"IOI 2025":{"familyId":"ioi-2025","label":"Ioi 2025","capability":"coding","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"Internal Data Science Tasks":{"familyId":"internal-data-science-tasks","label":"Internal Data Science Tasks","capability":"agentic","included":false,"reason":"Composite index or publisher-specific internal evaluation lacks a stable general-capability comparison unit."},"Internal Database Migration Tasks":{"familyId":"internal-database-migration-tasks","label":"Internal Database Migration Tasks","capability":"coding","included":false,"reason":"Composite index or publisher-specific internal evaluation lacks a stable general-capability comparison unit."},"Internal Design Tasks":{"familyId":"internal-design-tasks","label":"Internal Design Tasks","capability":"agentic","included":false,"reason":"Composite index or publisher-specific internal evaluation lacks a stable general-capability comparison unit."},"Internal Research Debugging Evaluation":{"familyId":"internal-research-debugging-evaluation","label":"Internal Research Debugging Evaluation","capability":"coding","included":false,"reason":"Composite index or publisher-specific internal evaluation lacks a stable general-capability comparison unit."},"JobBench":{"familyId":"jobbench","label":"Jobbench","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"KernelBench Hard":{"familyId":"kernelbench-hard","label":"Kernelbench Hard","capability":"coding","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"KernelGen 1P":{"familyId":"kernelgen-1p","label":"Kernelgen 1P","capability":"coding","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"Kimi Code Bench 2.0":{"familyId":"kimi-code-bench-2-0","label":"Kimi Code Bench 2 0","capability":"coding","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"LABBench":{"familyId":"labbench","label":"Labbench","capability":"knowledge","included":false,"reason":"Narrow specialist biological, medical or security evaluation; preserved as raw evidence outside general capability scoring."},"LOCA-Bench 256k":{"familyId":"loca-bench-256k","label":"Loca Bench 256K","capability":"long_context","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"LVBench":{"familyId":"lvbench","label":"Lvbench","capability":"multimodal","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"Legal Agent Benchmark":{"familyId":"harvey-lab","label":"Harvey Lab","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"Legal Research Bench":{"familyId":"legal-research-bench","label":"Legal Research Bench","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"LifeSciBench":{"familyId":"lifescibench","label":"Lifescibench","capability":"knowledge","included":false,"reason":"Narrow specialist biological, medical or security evaluation; preserved as raw evidence outside general capability scoring."},"LiveCodeBench":{"familyId":"livecodebench","label":"LiveCodeBench","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"LiveSQLBench":{"familyId":"livesqlbench","label":"Livesqlbench","capability":"coding","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"LongBench v2":{"familyId":"longbench","label":"Longbench","capability":"long_context","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"LongBenchV2":{"familyId":"longbench","label":"Longbench","capability":"long_context","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"LongCodeBench 1M":{"familyId":"longcodebench-1m","label":"Longcodebench 1M","capability":"long_context","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"Longbench v2 (≤ 1M)":{"familyId":"longbench","label":"Longbench","capability":"long_context","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"MCP-Atlas":{"familyId":"mcp-atlas","label":"MCP-Atlas","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"MCPAtlas":{"familyId":"mcp-atlas","label":"MCP-Atlas","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"MCPMark-Verified":{"familyId":"mcpmark-verified","label":"Mcpmark Verified","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"MILU":{"familyId":"milu","label":"Milu","capability":"knowledge","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"MLS-Bench-Lite":{"familyId":"mls-bench-lite","label":"Mls Bench Lite","capability":"coding","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"MMLU-Pro":{"familyId":"mmlu-pro","label":"MMLU-Pro","capability":"knowledge","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"MMLU-ProX (avg en/de/fr/es/it/ja/zh/hi/pt/ko)":{"familyId":"mmlu-pro","label":"MMLU-Pro","capability":"knowledge","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"MMMU":{"familyId":"mmmu","label":"Mmmu","capability":"multimodal","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"MMMU Pro (no tools)":{"familyId":"mmmu-pro-no-tools","label":"Mmmu Pro No Tools","capability":"multimodal","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"MMMU Pro (with tools)":{"familyId":"mmmu-pro-with-tools","label":"Mmmu Pro With Tools","capability":"multimodal","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"MMMU-Pro":{"familyId":"mmmu-pro","label":"Mmmu Pro","capability":"multimodal","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"MMVU":{"familyId":"mmvu","label":"Mmvu","capability":"multimodal","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"MRCR":{"familyId":"mrcr","label":"MRCR","capability":"long_context","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"MRCR v2 1M 8-needle":{"familyId":"mrcr","label":"MRCR","capability":"long_context","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"MRCR v2 256K (8-needle)":{"familyId":"mrcr","label":"MRCR","capability":"long_context","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"MT-AIME 2025 Arabic/Japanese/Korean":{"familyId":"mt-aime-2025-arabic-japanese-korean","label":"Mt Aime 2025 Arabic Japanese Korean","capability":"hard_reasoning","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"Management Consulting Tasks (Internal)":{"familyId":"management-consulting-tasks-internal","label":"Management Consulting Tasks Internal","capability":"agentic","included":false,"reason":"Composite index or publisher-specific internal evaluation lacks a stable general-capability comparison unit."},"MathVision":{"familyId":"mathvision","label":"Mathvision","capability":"multimodal","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"MathVista":{"familyId":"mathvista","label":"Mathvista","capability":"multimodal","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"MedChemBench (Internal)":{"familyId":"medchembench-internal","label":"Medchembench Internal","capability":"knowledge","included":false,"reason":"Narrow specialist biological, medical or security evaluation; preserved as raw evidence outside general capability scoring."},"MedXpertQA":{"familyId":"medxpertqa","label":"Medxpertqa","capability":"knowledge","included":false,"reason":"Narrow specialist biological, medical or security evaluation; preserved as raw evidence outside general capability scoring."},"Multi-Challenge":{"familyId":"multichallenge","label":"Multichallenge","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"MultiChallenge":{"familyId":"multichallenge","label":"Multichallenge","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"NL2Repo":{"familyId":"nl2repo","label":"Nl2Repo","capability":"coding","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"NL2Repo-Bench":{"familyId":"nl2repo-bench","label":"Nl2Repo Bench","capability":"coding","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"NanoGPT":{"familyId":"nanogpt","label":"Nanogpt","capability":"coding","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"No-CoT math time horizon":{"familyId":"no-cot-math-time-horizon","label":"No Cot Math Time Horizon","capability":"hard_reasoning","included":false,"reason":"Composite index or publisher-specific internal evaluation lacks a stable general-capability comparison unit."},"North Agentic Question Answering":{"familyId":"north-agentic-question-answering","label":"North Agentic Question Answering","capability":"agentic","included":false,"reason":"Composite index or publisher-specific internal evaluation lacks a stable general-capability comparison unit."},"North Data Analysis":{"familyId":"north-data-analysis","label":"North Data Analysis","capability":"agentic","included":false,"reason":"Composite index or publisher-specific internal evaluation lacks a stable general-capability comparison unit."},"North Memory Usage Quality":{"familyId":"north-memory-usage-quality","label":"North Memory Usage Quality","capability":"agentic","included":false,"reason":"Composite index or publisher-specific internal evaluation lacks a stable general-capability comparison unit."},"OCRBench v2 average accuracy":{"familyId":"ocrbench-v2-average-accuracy","label":"Ocrbench V2 Average Accuracy","capability":"multimodal","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"OSWorld":{"familyId":"osworld","label":"OSWorld","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"OfficeQA":{"familyId":"officeqa","label":"OfficeQA","capability":"agentic","included":true,"reason":"Agentic grounded reasoning over a provided document corpus. Pro shares one family budget; input extraction and tool configurations remain distinct. See https://github.com/databricks/officeqa."},"OfficeQA Pro":{"familyId":"officeqa","label":"OfficeQA","capability":"agentic","included":true,"reason":"Agentic grounded reasoning over a provided document corpus. Pro shares one family budget; input extraction and tool configurations remain distinct. See https://github.com/databricks/officeqa."},"OmniDocBench":{"familyId":"omnidocbench","label":"Omnidocbench","capability":"multimodal","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"OmniScience Accuracy":{"familyId":"omniscience-accuracy","label":"Omniscience Accuracy","capability":"knowledge","included":false,"reason":"Narrow specialist biological, medical or security evaluation; preserved as raw evidence outside general capability scoring."},"OpenAI MRCR v2 8-needle 256K-512K":{"familyId":"mrcr","label":"MRCR","capability":"long_context","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"OpenAI MRCR v2 8-needle 512K-1M":{"familyId":"mrcr","label":"MRCR","capability":"long_context","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"OpenScore String Quartets (1 - OMR-NED)":{"familyId":"openscore-string-quartets-1-omr-ned","label":"Openscore String Quartets 1 Omr Ned","capability":"multimodal","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"Organic Chemistry":{"familyId":"organic-chemistry","label":"Organic Chemistry","capability":"knowledge","included":false,"reason":"Narrow specialist biological, medical or security evaluation; preserved as raw evidence outside general capability scoring."},"PLawBench":{"familyId":"plawbench","label":"Plawbench","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"PRBench-Finance":{"familyId":"prbench-finance","label":"Prbench Finance","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"PRBench-Legal":{"familyId":"prbench-legal","label":"Prbench Legal","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"PaperBench":{"familyId":"paperbench","label":"Paperbench","capability":"coding","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"PerceptionBench":{"familyId":"perceptionbench","label":"Perceptionbench","capability":"multimodal","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"Phage-plasmid Co-evolution":{"familyId":"phage-plasmid-co-evolution","label":"Phage Plasmid Co Evolution","capability":"knowledge","included":false,"reason":"Narrow specialist biological, medical or security evaluation; preserved as raw evidence outside general capability scoring."},"PinchBench":{"familyId":"pinchbench","label":"Pinchbench","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"PostTrainBench":{"familyId":"posttrainbench","label":"Posttrainbench","capability":"coding","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"PostTrainBench Lite":{"familyId":"posttrainbench-lite","label":"Posttrainbench Lite","capability":"coding","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"ProfBench (Search)":{"familyId":"profbench-search","label":"Profbench Search","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"ProgramBench":{"familyId":"programbench","label":"Programbench","capability":"coding","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"ProgramBench (Almost Solved)":{"familyId":"programbench","label":"Programbench","capability":"coding","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"Protein Design":{"familyId":"protein-design","label":"Protein Design","capability":"knowledge","included":false,"reason":"Narrow specialist biological, medical or security evaluation; preserved as raw evidence outside general capability scoring."},"ProteinGym":{"familyId":"proteingym","label":"Proteingym","capability":"knowledge","included":false,"reason":"Narrow specialist biological, medical or security evaluation; preserved as raw evidence outside general capability scoring."},"ProtocolQA Open-Ended":{"familyId":"protocolqa-open-ended","label":"Protocolqa Open Ended","capability":"knowledge","included":false,"reason":"Narrow specialist biological, medical or security evaluation; preserved as raw evidence outside general capability scoring."},"Protocols":{"familyId":"protocols","label":"Protocols","capability":"knowledge","included":false,"reason":"Narrow specialist biological, medical or security evaluation; preserved as raw evidence outside general capability scoring."},"QVHighlights R1@0.5":{"familyId":"qvhighlights-r1-0-5","label":"Qvhighlights R1 0 5","capability":"multimodal","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"QwenQoderBench":{"familyId":"qwenqoderbench","label":"Qwenqoderbench","capability":"coding","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"QwenReactBench":{"familyId":"qwenreactbench","label":"Qwenreactbench","capability":"coding","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"QwenSVGBench":{"familyId":"qwensvgbench","label":"Qwensvgbench","capability":"coding","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"QwenSWEBench":{"familyId":"qwenswebench","label":"Qwenswebench","capability":"coding","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"RSI Index":{"familyId":"rsi-index","label":"Rsi Index","capability":"coding","included":false,"reason":"Composite index or publisher-specific internal evaluation lacks a stable general-capability comparison unit."},"RULER (1M)":{"familyId":"ruler-1m","label":"Ruler 1M","capability":"long_context","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"RealKIE-FCC Verified ALNS":{"familyId":"realkie-fcc-verified-alns","label":"Realkie Fcc Verified Alns","capability":"multimodal","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"ResearchRubrics":{"familyId":"researchrubrics","label":"Researchrubrics","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"SEC-Bench Pro":{"familyId":"sec-bench-pro","label":"Sec Bench Pro","capability":"coding","included":false,"reason":"Narrow specialist biological, medical or security evaluation; preserved as raw evidence outside general capability scoring."},"SHP2 Protein Function Prediction":{"familyId":"shp2-protein-function-prediction","label":"Shp2 Protein Function Prediction","capability":"knowledge","included":false,"reason":"Narrow specialist biological, medical or security evaluation; preserved as raw evidence outside general capability scoring."},"SRE-Bench":{"familyId":"sre-bench","label":"Sre Bench","capability":"coding","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"SVG-Bench":{"familyId":"svg-bench","label":"Svg Bench","capability":"coding","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"SWE-Atlas Codebase QnA":{"familyId":"swe-atlas-qna","label":"SWE-Atlas Codebase QnA","capability":"coding","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"SWE-Marathon":{"familyId":"swe-marathon","label":"SWE-Marathon","capability":"coding","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"SWE-Marathon (v1.1)":{"familyId":"swe-marathon","label":"SWE-Marathon","capability":"coding","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"SWE-bench Multilingual":{"familyId":"swe-bench-multilingual","label":"SWE-bench Multilingual","capability":"coding","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"SWE-bench Multimodal":{"familyId":"swe-bench-multimodal","label":"SWE-bench Multimodal","capability":"coding","included":true,"reason":"Repository patch correctness with visual issue context; software engineering remains the primary task."},"SWE-bench Pro":{"familyId":"swe-bench-pro","label":"SWE-bench Pro","capability":"coding","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"SWE-bench Verified":{"familyId":"swe-bench-verified","label":"SWE-bench Verified","capability":"coding","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"SWE-fficiency":{"familyId":"swe-fficiency","label":"Swe Fficiency","capability":"coding","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"SWEAtlas-QnA":{"familyId":"swe-atlas-qna","label":"SWE-Atlas Codebase QnA","capability":"coding","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"SWEAtlas-TestWriting":{"familyId":"sweatlas-testwriting","label":"Sweatlas Testwriting","capability":"coding","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"SaaS-Bench":{"familyId":"saas-bench","label":"Saas Bench","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"Sandbox Bench":{"familyId":"sandbox-bench","label":"Sandbox Bench","capability":"coding","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"SciCode":{"familyId":"scicode","label":"Scicode","capability":"coding","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"SciCode (subtask)":{"familyId":"scicode","label":"Scicode","capability":"coding","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"ScreenSpot point accuracy":{"familyId":"screenspot","label":"ScreenSpot","capability":"multimodal","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"ScreenSpot-Pro":{"familyId":"screenspot","label":"ScreenSpot","capability":"multimodal","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"SimpleQA Verified":{"familyId":"simpleqa-verified","label":"Simpleqa Verified","capability":"knowledge","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"SingleCellBench":{"familyId":"singlecellbench","label":"Singlecellbench","capability":"knowledge","included":false,"reason":"Narrow specialist biological, medical or security evaluation; preserved as raw evidence outside general capability scoring."},"SkillsBench":{"familyId":"skillsbench","label":"Skillsbench","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"SpatialBench":{"familyId":"spatialbench","label":"Spatialbench","capability":"knowledge","included":false,"reason":"Narrow specialist biological, medical or security evaluation; preserved as raw evidence outside general capability scoring."},"SpreadsheetBench 2":{"familyId":"spreadsheetbench-2","label":"Spreadsheetbench 2","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"SpreadsheetBench v1":{"familyId":"spreadsheetbench-v1","label":"Spreadsheetbench V1","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"Tacit Knowledge and Troubleshooting":{"familyId":"tacit-knowledge-and-troubleshooting","label":"Tacit Knowledge And Troubleshooting","capability":"knowledge","included":false,"reason":"Narrow specialist biological, medical or security evaluation; preserved as raw evidence outside general capability scoring."},"TauBench V3 Airline":{"familyId":"tau-bench","label":"Tau Bench","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"TauBench V3 Average":{"familyId":"tau-bench","label":"Tau Bench","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"TauBench V3 Banking":{"familyId":"tau-bench","label":"Tau Bench","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"TauBench V3 Retail":{"familyId":"tau-bench","label":"Tau Bench","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"TauBench V3 Telecom":{"familyId":"tau-bench","label":"Tau Bench","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"Terminal-Bench":{"familyId":"terminal-bench","label":"Terminal-Bench","capability":"coding","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"Toolathlon":{"familyId":"toolathlon","label":"Toolathlon","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"Toolathlon Verified":{"familyId":"toolathlon","label":"Toolathlon","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"Toolathlon Verified (Pass@1)":{"familyId":"toolathlon","label":"Toolathlon","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"Toolathlon-Verified":{"familyId":"toolathlon","label":"Toolathlon","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"TroubleshootingBench":{"familyId":"troubleshootingbench","label":"Troubleshootingbench","capability":"knowledge","included":false,"reason":"Narrow specialist biological, medical or security evaluation; preserved as raw evidence outside general capability scoring."},"USAMO 2026":{"familyId":"usamo-2026","label":"Usamo 2026","capability":"hard_reasoning","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"VIBE-V2":{"familyId":"vibe-v2","label":"Vibe V2","capability":"coding","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"Vals.ai Financial Agent 1.1 with web search":{"familyId":"finance-agent","label":"Finance Agent","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"Vals.ai Financial Agent 1.1 without web search":{"familyId":"finance-agent","label":"Finance Agent","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"Video-MME (w. sub)":{"familyId":"videomme","label":"Video-MME","capability":"multimodal","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"VideoMME with subtitles":{"familyId":"videomme","label":"Video-MME","capability":"multimodal","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"VideoMMMU":{"familyId":"videommmu","label":"Videommmu","capability":"multimodal","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"WMT24++ (en→xx)":{"familyId":"wmt24","label":"WMT24++","capability":"knowledge","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"WMT24++ 50 varieties":{"familyId":"wmt24","label":"WMT24++","capability":"knowledge","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"WebArena Verified":{"familyId":"webarena-verified","label":"Webarena Verified","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"WideSearch":{"familyId":"widesearch","label":"Widesearch","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"WorkSpaceBench":{"familyId":"workspacebench","label":"Workspacebench","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"WorldVQA ForceAnswer":{"familyId":"worldvqa-forceanswer","label":"Worldvqa Forceanswer","capability":"multimodal","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"YC-Bench":{"familyId":"yc-bench","label":"Yc Bench","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"ZeroBench (pass@5)":{"familyId":"zerobench-pass-5","label":"Zerobench Pass 5","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"gdp.pdf":{"familyId":"gdp-pdf","label":"Gdp Pdf","capability":"multimodal","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"tau2 Airline Verified":{"familyId":"tau-bench","label":"Tau Bench","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"tau2 Retail Verified":{"familyId":"tau-bench","label":"Tau Bench","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"tau2 Telecom":{"familyId":"tau-bench","label":"Tau Bench","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"tau2-Bench Telecom":{"familyId":"tau-bench","label":"Tau Bench","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"tau3 Airline":{"familyId":"tau-bench","label":"Tau Bench","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"tau3 Banking":{"familyId":"tau-bench","label":"Tau Bench","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"tau3 Retail":{"familyId":"tau-bench","label":"Tau Bench","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"tau3 Telecom":{"familyId":"tau-bench","label":"Tau Bench","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"τ³-Banking":{"familyId":"banking","label":"Banking","capability":"agentic","included":true,"reason":"Direct performance benchmark; source/version/metric/configuration matching remains required."},"swe-bench-verified":{"familyId":"swe-bench-verified","label":"SWE-bench Verified","capability":"coding","included":true,"reason":"Original benchmark; same direct matched-unit rules apply. Human preference is reserved for direct human preference evidence."},"lmarena-text":{"familyId":"human-preference","label":"Human preference (Arena)","capability":"human_pref","included":true,"reason":"Original benchmark; same direct matched-unit rules apply. Human preference is reserved for direct human preference evidence."},"gdpval-aa":{"familyId":"gdpval","label":"GDPval","capability":"agentic","included":true,"reason":"Original benchmark; same direct matched-unit rules apply. Human preference is reserved for direct human preference evidence."},"terminal-bench-2.1":{"familyId":"terminal-bench","label":"Terminal-Bench","capability":"coding","included":true,"reason":"Original benchmark; same direct matched-unit rules apply. Human preference is reserved for direct human preference evidence."},"swe-bench-pro":{"familyId":"swe-bench-pro","label":"SWE-bench Pro","capability":"coding","included":true,"reason":"Original benchmark; same direct matched-unit rules apply. Human preference is reserved for direct human preference evidence."},"hle":{"familyId":"hle","label":"Humanity’s Last Exam","capability":"hard_reasoning","included":true,"reason":"Original benchmark; same direct matched-unit rules apply. Human preference is reserved for direct human preference evidence."},"arc-agi-2":{"familyId":"arc-agi","label":"ARC-AGI","capability":"hard_reasoning","included":true,"reason":"ARC generations share one family weight; task versions remain separate comparison units and cannot be paired across generations."},"osworld-verified":{"familyId":"osworld","label":"OSWorld","capability":"agentic","included":true,"reason":"Original benchmark; same direct matched-unit rules apply. Human preference is reserved for direct human preference evidence."},"livecodebench":{"familyId":"livecodebench","label":"LiveCodeBench","capability":"coding","included":true,"reason":"Original benchmark; same direct matched-unit rules apply. Human preference is reserved for direct human preference evidence."},"deepswe-v1.1":{"familyId":"deepswe","label":"DeepSWE","capability":"coding","included":true,"reason":"Original benchmark; same direct matched-unit rules apply. Human preference is reserved for direct human preference evidence."},"gpqa-diamond":{"familyId":"gpqa","label":"GPQA","capability":"hard_reasoning","included":true,"reason":"Original benchmark; same direct matched-unit rules apply. Human preference is reserved for direct human preference evidence."},"mmlu-pro":{"familyId":"mmlu-pro","label":"MMLU-Pro","capability":"knowledge","included":true,"reason":"Original benchmark; same direct matched-unit rules apply. Human preference is reserved for direct human preference evidence."},"aa-intelligence-index":{"familyId":"aa-intelligence-index","label":"Aa Intelligence Index","capability":"knowledge","included":false,"reason":"Composite index excluded to avoid double counting its ingredients."},"browsecomp":{"familyId":"browsecomp","label":"Browsecomp","capability":"agentic","included":true,"reason":"Original benchmark; direct matched-unit rules apply."},"tau2-telecom":{"familyId":"tau-bench","label":"Tau Bench","capability":"agentic","included":true,"reason":"Original benchmark; direct matched-unit rules apply."},"paperbench":{"familyId":"paperbench","label":"Paperbench","capability":"coding","included":true,"reason":"Original benchmark; direct matched-unit rules apply."},"Terminal-Bench Science":{"familyId":"terminal-bench-science","label":"Terminal-Bench Science","capability":"hard_reasoning","included":true,"reason":"Agentic scientific reasoning tasks, distinct from terminal coding; match source, version and settings."}}},"scoring":{"version":"2026-09-07-v4-core-4","coreFamilies":{"coding":["swe-bench-verified","swe-bench-pro","swe-bench-multilingual","deepswe","livecodebench","terminal-bench","scicode"],"hard_reasoning":["hle","gpqa","arc-agi","aime","frontiermath","imo-answerbench","terminal-bench-science"],"agentic":["osworld","webarena-verified","browsecomp","toolathlon","mcp-atlas","automationbench","deepsearchqa","gdpval","bfcl-v4","apex-agents","officeqa"],"knowledge":["mmlu-pro","gmmlu","milu","simpleqa-verified"],"multimodal":["mmmu","mmmu-pro","charxiv","mathvista","chartography","lvbench","screenspot","ocrbench-v2-average-accuracy"],"long_context":["mrcr","longbench","ruler-1m","graphwalks","aa-lcr"],"human_pref":["human-preference"]},"metricAdmission":{"allowedUnits":["%","percent","percent F1","percent sequence match"],"familyAllowedUnits":{"human-preference":["elo","Elo"],"graphwalks":["%","percent","percent F1"],"mrcr":["%","percent","percent sequence match"],"gdpval":["elo","Elo","percent","%"]},"excludedUnits":{"turns":"Efficiency is not task success.","USD":"Cost is not task success.","minutes":"Latency is not task success.","count":"Requires a reviewed denominator and task-success metric.","index":"Unvalidated composite scale.","index score":"Unvalidated composite scale.","index points":"Unvalidated composite scale.","composite score":"Unvalidated composite scale.","rating":"Model-judged rating is not direct human preference.","score (source scale)":"Unspecified scale lacks a validated task-success definition."},"unknownUnitPolicy":"exclude"},"excludedObservationIds":{"launch-anthropic-fable-5-1-card-85":"System card section8.15.5 confirms11of324trials were partly or fully completed by Opus4.8fallback; actual mixed-model execution, not merely available fallback.","launch-anthropic-fable-5-1-card-87":"System card section8.15.5 confirms11of324trials were partly or fully completed by Opus4.8fallback; actual mixed-model execution, not merely available fallback.","launch-anthropic-fable-5-1-card-89":"System card section8.15.5 confirms11of324trials were partly or fully completed by Opus4.8fallback; actual mixed-model execution, not merely available fallback.","launch-anthropic-fable-5-1-card-91":"System card section8.15.5 confirms11of324trials were partly or fully completed by Opus4.8fallback; actual mixed-model execution, not merely available fallback.","launch-meta-muse-1-3-report-2035":"Different native coding harnesses are mixed; exact shared agent configuration is not established.","launch-meta-muse-1-3-report-2036":"Different native coding harnesses are mixed; exact shared agent configuration is not established.","launch-meta-muse-1-3-report-2037":"Different native coding harnesses are mixed; exact shared agent configuration is not established.","launch-meta-muse-1-3-report-2038":"Different native coding harnesses are mixed; exact shared agent configuration is not established.","launch-meta-muse-1-3-report-2039":"Different native coding harnesses are mixed; exact shared agent configuration is not established.","launch-google-gemini-3-8-card-2081":"Card reports agentic result but linked methodology describes only static no-tools protocol.","launch-google-gemini-3-8-card-2082":"Mixed source task revisions and best-of-three versus provider reporting prevent a uniform common-core unit.","launch-google-gemini-3-8-card-2083":"Mixed source task revisions and best-of-three versus provider reporting prevent a uniform common-core unit.","launch-google-gemini-3-8-card-2084":"Mixed source task revisions and best-of-three versus provider reporting prevent a uniform common-core unit.","launch-google-gemini-3-8-card-2085":"Mixed source task revisions and best-of-three versus provider reporting prevent a uniform common-core unit.","launch-google-gemini-3-8-card-2086":"Provider-sourced OSWorld task revision is unverified; cannot join a known-version comparison.","launch-google-gemini-3-8-card-2087":"Provider-sourced OSWorld task revision is unverified; cannot join a known-version comparison.","launch-microsoft-1392":"Microsoft report Table 11 and section 4.1: MAI Terminal-Bench removes timeouts and uses a minimal ReAct harness; comparator values are cited from other model releases. A common evaluation protocol is not established.","launch-microsoft-1393":"Microsoft report Table 11 and section 4.1: MAI Terminal-Bench removes timeouts and uses a minimal ReAct harness; comparator values are cited from other model releases. A common evaluation protocol is not established.","launch-microsoft-1394":"Microsoft report Table 11 and section 4.1: MAI Terminal-Bench removes timeouts and uses a minimal ReAct harness; comparator values are cited from other model releases. A common evaluation protocol is not established.","launch-microsoft-1395":"Microsoft report Table 11 and section 4.1: MAI Terminal-Bench removes timeouts and uses a minimal ReAct harness; comparator values are cited from other model releases. A common evaluation protocol is not established.","launch-microsoft-1396":"Microsoft report Table 11 and section 4.1: MAI Terminal-Bench removes timeouts and uses a minimal ReAct harness; comparator values are cited from other model releases. A common evaluation protocol is not established.","launch-microsoft-1397":"Microsoft report Table 11 and section 4.1: MAI Terminal-Bench removes timeouts and uses a minimal ReAct harness; comparator values are cited from other model releases. A common evaluation protocol is not established.","launch-minimax-1708":"MiniMax cites this comparator from the official Terminal-Bench leaderboard; a common protocol with its internal Terminus 2 run is not established. See MiniMax M3 benchmark figure, Terminal-bench 2.1 methodology.","launch-minimax-1709":"MiniMax cites this comparator from the official Terminal-Bench leaderboard; a common protocol with its internal Terminus 2 run is not established. See MiniMax M3 benchmark figure, Terminal-bench 2.1 methodology.","launch-minimax-1710":"MiniMax cites this comparator from the official Terminal-Bench leaderboard; a common protocol with its internal Terminus 2 run is not established. See MiniMax M3 benchmark figure, Terminal-bench 2.1 methodology."},"excludedObservationFingerprints":{"launch-anthropic-fable-5-1-card-85":{"modelId":"claude-fable-5-1","benchmarkName":"Toolathlon","benchmarkVersion":"Verified June2026","score":77.8,"unit":"percent","configuration":"Pass@1;108tasks/three trials; internal harness; max effort; Fable safeguards+Opus4.8fallback; Opus safeguards/fallback disabled; pinned containers/data.","sourceId":"anthropic-fable-5-1-card"},"launch-anthropic-fable-5-1-card-87":{"modelId":"claude-fable-5-1","benchmarkName":"Toolathlon","benchmarkVersion":"Verified June2026","score":81.5,"unit":"percent","configuration":"Pass@3;108tasks/three trials; internal harness; max effort; Fable safeguards+Opus4.8fallback; Opus safeguards/fallback disabled; pinned containers/data.","sourceId":"anthropic-fable-5-1-card"},"launch-anthropic-fable-5-1-card-89":{"modelId":"claude-fable-5-1","benchmarkName":"Toolathlon","benchmarkVersion":"Verified June2026","score":73.1,"unit":"percent","configuration":"Pass³;108tasks/three trials; internal harness; max effort; Fable safeguards+Opus4.8fallback; Opus safeguards/fallback disabled; pinned containers/data.","sourceId":"anthropic-fable-5-1-card"},"launch-anthropic-fable-5-1-card-91":{"modelId":"claude-fable-5-1","benchmarkName":"Toolathlon","benchmarkVersion":"Verified June2026","score":23.7,"unit":"turns","configuration":"average turns;108tasks/three trials; internal harness; max effort; Fable safeguards+Opus4.8fallback; Opus safeguards/fallback disabled; pinned containers/data.","sourceId":"anthropic-fable-5-1-card"},"launch-meta-muse-1-3-report-2035":{"modelId":"muse-spark-1.3","benchmarkName":"Terminal-Bench","benchmarkVersion":"2.1","score":88.8,"unit":"percent","configuration":"89tasks; each model nativecoding harness in internalframework/cloudsandbox; officialverifier; GPT owncard; reasoning max; native harness family: Meta (exact harness revision not specified)","sourceId":"meta-muse-1-3-report"},"launch-meta-muse-1-3-report-2036":{"modelId":"muse-spark-1.3","benchmarkName":"Terminal-Bench","benchmarkVersion":"2.1","score":89.2,"unit":"percent","configuration":"89tasks; each model nativecoding harness in internalframework/cloudsandbox; officialverifier; GPT owncard; reasoning xhigh; native harness family: Meta (exact harness revision not specified)","sourceId":"meta-muse-1-3-report"},"launch-meta-muse-1-3-report-2037":{"modelId":"muse-spark-1.2","benchmarkName":"Terminal-Bench","benchmarkVersion":"2.1","score":82.9,"unit":"percent","configuration":"89tasks; each model nativecoding harness in internalframework/cloudsandbox; officialverifier; GPT owncard; reasoning xhigh; native harness family: Meta (exact harness revision not specified)","sourceId":"meta-muse-1-3-report"},"launch-meta-muse-1-3-report-2038":{"modelId":"gpt-5.6-sol","benchmarkName":"Terminal-Bench","benchmarkVersion":"2.1","score":88.8,"unit":"percent","configuration":"89tasks; each model nativecoding harness in internalframework/cloudsandbox; officialverifier; GPT owncard; reasoning max; native harness family: OpenAI (exact harness revision not specified)","sourceId":"meta-muse-1-3-report"},"launch-meta-muse-1-3-report-2039":{"modelId":"claude-opus-5","benchmarkName":"Terminal-Bench","benchmarkVersion":"2.1","score":86.7,"unit":"percent","configuration":"89tasks; each model nativecoding harness in internalframework/cloudsandbox; officialverifier; GPT owncard; reasoning max; native harness family: Anthropic (exact harness revision not specified)","sourceId":"meta-muse-1-3-report"},"launch-google-gemini-3-8-card-2081":{"modelId":"gemini-3.8-flash","benchmarkName":"LVBench","benchmarkVersion":"agentic","score":87.8,"unit":"percent","configuration":"Card labels agentic; linkedmethodology describes only no-tools static setup","sourceId":"google-gemini-3-8-card"},"launch-google-gemini-3-8-card-2082":{"modelId":"gemini-3.8-flash","benchmarkName":"OSWorld partial","benchmarkVersion":"2.0 pre-08.08","score":59,"unit":"percent","configuration":"Partialscore; batchtools;1080p/500steps; Gemini/Sonnet bestof3runs; screenshotonly; officialCUAharness","sourceId":"google-gemini-3-8-card"},"launch-google-gemini-3-8-card-2083":{"modelId":"gemini-3.7-flash","benchmarkName":"OSWorld partial","benchmarkVersion":"2.0 pre-08.08","score":50.6,"unit":"percent","configuration":"Partialscore; batchtools;1080p/500steps; Gemini/Sonnet bestof3runs; screenshotonly; officialCUAharness","sourceId":"google-gemini-3-8-card"},"launch-google-gemini-3-8-card-2084":{"modelId":"claude-opus-5","benchmarkName":"OSWorld partial","benchmarkVersion":"2.0 August2026 fixed tasks","score":75.4,"unit":"percent","configuration":"Partialscore; batchtools;1080p/500steps; Gemini/Sonnet bestof3runs; screenshotonly; officialCUAharness","sourceId":"google-gemini-3-8-card"},"launch-google-gemini-3-8-card-2085":{"modelId":"claude-sonnet-5","benchmarkName":"OSWorld partial","benchmarkVersion":"2.0 pre-08.08","score":42.6,"unit":"percent","configuration":"Partialscore; batchtools;1080p/500steps; Gemini/Sonnet bestof3runs; screenshotonly; officialCUAharness","sourceId":"google-gemini-3-8-card"},"launch-google-gemini-3-8-card-2086":{"modelId":"gpt-5.6-sol","benchmarkName":"OSWorld partial","benchmarkVersion":"2.0 task revision unverified gpt-5.6-sol","score":62.6,"unit":"percent","configuration":"Partialscore; batchtools;1080p/500steps; Gemini/Sonnet bestof3runs; screenshotonly; officialCUAharness","sourceId":"google-gemini-3-8-card"},"launch-google-gemini-3-8-card-2087":{"modelId":"gpt-5.6-terra","benchmarkName":"OSWorld partial","benchmarkVersion":"2.0 task revision unverified gpt-5.6-terra","score":50.2,"unit":"percent","configuration":"Partialscore; batchtools;1080p/500steps; Gemini/Sonnet bestof3runs; screenshotonly; officialCUAharness","sourceId":"google-gemini-3-8-card"},"launch-microsoft-1392":{"modelId":"mai-thinking-1","benchmarkName":"Terminal-Bench 2.0","benchmarkVersion":"2.0","score":46,"unit":"%","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports.","sourceId":"microsoft"},"launch-microsoft-1393":{"modelId":"claude-sonnet-4-6","benchmarkName":"Terminal-Bench 2.0","benchmarkVersion":"2.0","score":59.1,"unit":"%","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports.","sourceId":"microsoft"},"launch-microsoft-1394":{"modelId":"claude-opus-4-6","benchmarkName":"Terminal-Bench 2.0","benchmarkVersion":"2.0","score":65.4,"unit":"%","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports.","sourceId":"microsoft"},"launch-microsoft-1395":{"modelId":"gpt-5.4","benchmarkName":"Terminal-Bench 2.0","benchmarkVersion":"2.0","score":75.1,"unit":"%","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports.","sourceId":"microsoft"},"launch-microsoft-1396":{"modelId":"kimi-k2.6","benchmarkName":"Terminal-Bench 2.0","benchmarkVersion":"2.0","score":66.7,"unit":"%","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports.","sourceId":"microsoft"},"launch-microsoft-1397":{"modelId":"glm-5.1","benchmarkName":"Terminal-Bench 2.0","benchmarkVersion":"2.0","score":69,"unit":"%","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports.","sourceId":"microsoft"},"launch-minimax-1708":{"modelId":"claude-opus-4-7","benchmarkName":"Terminal-Bench 2.1","benchmarkVersion":"2.1","score":66.1,"unit":"%","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","sourceId":"minimax"},"launch-minimax-1709":{"modelId":"gpt-5.5","benchmarkName":"Terminal-Bench 2.1","benchmarkVersion":"2.1","score":78.2,"unit":"%","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","sourceId":"minimax"},"launch-minimax-1710":{"modelId":"gemini-3.1-pro","benchmarkName":"Terminal-Bench 2.1","benchmarkVersion":"2.1","score":70.3,"unit":"%","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","sourceId":"minimax"}},"matchingRules":{"sameSource":true,"sameVersion":true,"sameMetric":true,"sameConfiguration":true,"excludeComparableFalse":true,"familyBalanced":true,"allowDifferentSettingsOverride":false,"unknownModelIdentity":"exclude","mixedModelFallback":"exclude"},"familyAliases":{"mmmu-pro-no-tools":"mmmu-pro","mmmu-pro-with-tools":"mmmu-pro"},"notes":["Admission identifies a common-core candidate, not proof of controlled comparability or equal inference compute.","Fallback, specialist and unsupported metric observations remain visible as raw evidence.","Human preference core has one defensible family; no minimum family count is manufactured.","Observation IDs follow current launchEvidence concatenation order; apply fingerprint checks to detect changed identity after source edits.","The same three reference products are used across all capabilities. Opus 4.8 has connected long-context observations; Opus 5 does not in this audited snapshot. Panel chosen for cross-capability coverage, not output ordering."],"referencePanels":{"coding":["gpt-5.6-sol","claude-opus-4-8","gemini-3.1-pro"],"hard_reasoning":["gpt-5.6-sol","claude-opus-4-8","gemini-3.1-pro"],"agentic":["gpt-5.6-sol","claude-opus-4-8","gemini-3.1-pro"],"knowledge":["gpt-5.6-sol","claude-opus-4-8","gemini-3.1-pro"],"multimodal":["gpt-5.6-sol","claude-opus-4-8","gemini-3.1-pro"],"long_context":["gpt-5.6-sol","claude-opus-4-8","gemini-3.1-pro"],"human_pref":["gpt-5.6-sol","claude-opus-4-8","gemini-3.1-pro"]},"defaultWeights":{"agentic":0.34,"hard_reasoning":0.33,"coding":0.33,"human_pref":0,"knowledge":0,"multimodal":0,"long_context":0},"outcomeSmoothing":0.2,"smoothingChecks":[0.1,0.2,0.4],"evaluationOrigins":[{"benchmarkPrefix":"GDPval-AA","metric":"Elo","origin":"artificialanalysis.ai","evidenceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","note":"GDPval-AA is conducted by Artificial Analysis. Republishing the rating does not create a new evaluation source or upgrade a secondhand report to independent evidence."}],"taskReview":{"officeqa":{"task":"Agentic grounded document reasoning","decision":"Moved from knowledge to agentic; requires operating over a provided document corpus, not closed-book recall.","sourceUrl":"https://github.com/databricks/officeqa"},"ruler-1m":{"task":"Synthetic long-context retrieval, tracing, aggregation and QA","decision":"Keep one family; the aggregate does not supply separable subtask observations.","sourceUrl":"https://github.com/NVIDIA/RULER"},"longbench":{"task":"Realistic long-context comprehension and reasoning","decision":"Keep separate from RULER; shared capability membership is not proof of task interchangeability.","sourceUrl":"https://github.com/THUDM/LongBench"}},"provenanceReview":{"reviewedOn":"2026-09-07","sources":[{"sourceId":"minimax","decision":"Exclude only the three externally sourced Terminal-Bench comparators from matched scoring; retain internally evaluated opponents and all raw observations.","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","locator":"figures/benchmark.jpeg, Terminal-bench 2.1 methodology"},{"sourceId":"zai","decision":"GDPval-AA belongs to Artificial Analysis. Other admitted rows retain Z.ai as evaluator where benchmark footnotes describe its runs.","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3","locator":"Footnotes"},{"sourceId":"nvidia","decision":"Retain NVIDIA as evaluator: the model card states that all table results were collected through NeMo Evaluator SDK. No inference of copying from equal values.","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","locator":"Paragraphs immediately below benchmark table"}],"limits":"Origin groups identify evaluators, not exact run identities. Unknown origins use the publishing source. Historical board snapshots stay separate; matching values alone do not prove copying."}},"sourcePublishers":{"anthropic-fable-5-1-card":"Anthropic","openai-astra-launch":"OpenAI","openai-sol-launch":"OpenAI","openai-astra-system-card":"OpenAI","qwen":"Alibaba","kimi":"Moonshot AI","kimi-k26-card":"Moonshot AI","nvidia":"NVIDIA","deepseek":"DeepSeek","zai":"Z.ai","xai":"xAI","google":"Google","microsoft":"Microsoft","cohere":"Cohere","mistral":"Mistral","amazon":"Amazon","minimax":"MiniMax","meta":"Meta","meta-muse-1-3-report":"Meta","google-gemini-3-8-card":"Google"},"secondaryEvaluationReview":{"reviewedOn":"2026-09-07","rules":[{"benchmarkPrefix":"GDPval-AA","benchmarkVersion":"2","metric":"elo","primaryEvaluationId":"aa-gdpval-aa","primarySource":"Artificial Analysis","reason":"Secondary GDPval-AA v2 leaderboard report; the original evaluator already supplies this exact configuration. Reprinted or rerated board snapshots are not additional evaluation evidence.","evidenceUrls":["https://artificialanalysis.ai/evaluations/gdpval-aa","https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"],"locator":"Artificial Analysis Methodology; Anthropic system card section 8.15.3","limits":"Applies only when the reviewed primary unit contains the exact configuration. Distinct task versions, efforts, independently rerun experiments and unreviewed lineages are not deduplicated by score similarity."}]},"defaultWeights":{"agentic":0.34,"hard_reasoning":0.33,"coding":0.33,"human_pref":0,"knowledge":0,"multimodal":0,"long_context":0},"implementation":"Joint configuration comparisons with per-configuration evaluator budgets, then family, evaluation and opponent allocation. Each edge uses the smaller endpoint allocation; identical evidence profiles share weight. Fixed configuration reference panel."}