{"benchmarkAliases":[{"alias":"swe-verified","benchmarkId":"swe-bench-verified"},{"alias":"tb2.1","benchmarkId":"terminal-bench-2.1"},{"alias":"tb4.0","benchmarkId":"terminal-bench-4.0"},{"alias":"gpqa","benchmarkId":"gpqa-diamond"},{"alias":"aa-index","benchmarkId":"aa-intelligence-index"}],"benchmarks":[{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"vulcanbench-v3","name":"VulcanBench v3","preferredHarnessId":"vulcanbench-v3-07-grok-fable-sol-high","scoreUnit":"percent","url":"https://vulcanbench.com/benchmarks.html","version":"v3"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"autoresearchexam","name":"AutoResearchExam","preferredHarnessId":"autoresearchexam-terminus-2","scoreUnit":"index","url":"https://benchmarks.bespokelabs.ai/autoresearchexam/","version":"2026-09-09"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"bughunt-bench","name":"Bug Hunt Bench","preferredHarnessId":"bughunt-bench-max","scoreUnit":"percent","url":"https://bughunt.productcompass.pm/?preset=featured","version":"2026-09-22"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"real-swe","name":"Real-SWE","preferredHarnessId":"real-swe-fable-5-1-claude-code","scoreUnit":"percent","url":"https://realswe.withspecific.com/","version":"2026-09"},{"combiner":{"bucket":"agentic","inDefault":true},"higherIsBetter":true,"id":"swe-bench-verified","name":"SWE-bench Verified","preferredHarnessId":"swe-bench-verified-official","scoreUnit":"percent","url":"https://www.swebench.com/","version":"verified"},{"combiner":{"bucket":"agentic","inDefault":true},"higherIsBetter":true,"id":"swe-bench-pro","name":"SWE-bench Pro","preferredHarnessId":"swe-bench-pro-official","scoreUnit":"percent","url":"https://scale.com/leaderboard/swe_bench_pro_public","version":"public"},{"combiner":{"bucket":"agentic","inDefault":true},"higherIsBetter":true,"id":"terminal-bench-2.1","name":"Terminal-Bench 2.1","preferredHarnessId":"terminal-bench-2.1-reported","scoreUnit":"percent","url":"https://www.tbench.ai/?version=2.1","version":"2.1"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"terminal-bench-4.0","name":"Terminal-Bench 4.0","preferredHarnessId":"terminal-bench-4.0-reported","scoreUnit":"percent","url":"https://www.tbench.ai/","version":"4.0"},{"combiner":{"bucket":"agentic","inDefault":true},"higherIsBetter":true,"id":"deepswe-v1.1","name":"DeepSWE v1.1","preferredHarnessId":"deepswe-v1.1-reported","scoreUnit":"percent","url":"https://deepswe.datacurve.ai/","version":"1.1"},{"combiner":{"bucket":"agentic","inDefault":true},"higherIsBetter":true,"id":"osworld-verified","name":"OSWorld-Verified","preferredHarnessId":"osworld-verified-reported","scoreUnit":"percent","url":"https://os-world.github.io/","version":"verified"},{"combiner":{"bucket":"agentic","inDefault":true},"higherIsBetter":true,"id":"gdpval-aa","name":"GDPval-AA","preferredHarnessId":"gdpval-aa-official","scoreUnit":"elo","url":"https://artificialanalysis.ai/evaluations/gdpval-aa","version":"v2"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"automationbench-aa","name":"AutomationBench-AA","preferredHarnessId":"automationbench-aa-official","scoreUnit":"percent","url":"https://artificialanalysis.ai/evaluations/automationbench-aa","version":"aa-score"},{"combiner":{"bucket":"hard_reasoning","inDefault":true},"higherIsBetter":true,"id":"gpqa-diamond","name":"GPQA Diamond","preferredHarnessId":"gpqa-diamond-reported","scoreUnit":"percent","url":"https://github.com/idavidrein/gpqa","version":"diamond"},{"combiner":{"bucket":"hard_reasoning","inDefault":true},"higherIsBetter":true,"id":"hle","name":"Humanity's Last Exam","preferredHarnessId":"hle-no-tools","scoreUnit":"percent","url":"https://lastexam.ai/","version":"full"},{"combiner":{"bucket":"hard_reasoning","inDefault":true},"higherIsBetter":true,"id":"arc-agi-2","name":"ARC-AGI-2","preferredHarnessId":"arc-agi-2-official","scoreUnit":"percent","url":"https://arcprize.org/leaderboard","version":"2"},{"combiner":{"bucket":"coding","inDefault":true},"higherIsBetter":true,"id":"livecodebench","name":"LiveCodeBench","preferredHarnessId":"livecodebench-reported","scoreUnit":"percent","url":"https://livecodebench.github.io/","version":"v6"},{"combiner":{"bucket":"human_pref","inDefault":true},"higherIsBetter":true,"id":"lmarena-text","name":"LMArena Text Arena","preferredHarnessId":"lmarena-text-official","scoreUnit":"elo","url":"https://lmarena.ai/leaderboard","version":null},{"combiner":{"bucket":"knowledge","inDefault":true},"higherIsBetter":true,"id":"mmlu-pro","name":"MMLU-Pro","preferredHarnessId":"mmlu-pro-reported","scoreUnit":"percent","url":"https://github.com/TIGER-AI-Lab/MMLU-Pro","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"aa-intelligence-index","name":"Artificial Analysis Intelligence Index","preferredHarnessId":"aa-intelligence-index-official","scoreUnit":"index","url":"https://artificialanalysis.ai/","version":"v4.1"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"browsecomp","name":"BrowseComp","preferredHarnessId":"browsecomp-reported","scoreUnit":"percent","url":"https://openai.com/index/gpt-5-6/","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"tau2-telecom","name":"τ²-bench Telecom","preferredHarnessId":"tau2-telecom-reported","scoreUnit":"percent","url":"https://github.com/sierra-research/tau-bench","version":"telecom"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"paperbench","name":"PaperBench","preferredHarnessId":"paperbench-reported","scoreUnit":"percent","url":"https://openai.com/index/gpt-5-6/","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"swe-bench-pro-percent-higher","name":"SWE-bench Pro","preferredHarnessId":"swe-bench-pro-percent-higher:anthropic-fable-5-1-card:QW50aHJvcGljIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IGFkYXB0aXZlIHRoaW5raW5nIGF0IG1heCBlZmZvcnQsIGRlZmF1bHQgc2FtcGxpbmcsIGZpdmUtdHJpYWwgbWVhbiB1bmxlc3Mgc2VjdGlvbiBzcGVjaWZpZXMgb3RoZXJ3aXNlOyBjb250ZXh0IGF0IG1vc3QgMU0uIENvbXBhcmF0b3IgY29uZmlndXJhdGlvbiBmb2xsb3dzIGNpdGVkIHByaW9yIGNhcmQgb3IgYm9hcmQuIEFnZW50IGlkZW50aXR5IGlzIG5vdCBlc3RhYmxpc2hlZCBhcyBtaW5pLXN3ZS1hZ2VudC4","scoreUnit":"percent","url":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"swe-bench-multilingual-percent","name":"SWE-bench Multilingual","preferredHarnessId":"swe-bench-multilingual-percent:anthropic-fable-5-1-card:QW50aHJvcGljIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IGFkYXB0aXZlIHRoaW5raW5nIGF0IG1heCBlZmZvcnQsIGRlZmF1bHQgc2FtcGxpbmcsIGZpdmUtdHJpYWwgbWVhbiB1bmxlc3Mgc2VjdGlvbiBzcGVjaWZpZXMgb3RoZXJ3aXNlOyBjb250ZXh0IGF0IG1vc3QgMU0uIENvbXBhcmF0b3IgY29uZmlndXJhdGlvbiBmb2xsb3dzIGNpdGVkIHByaW9yIGNhcmQgb3IgYm9hcmQuIEFnZW50IGlkZW50aXR5IGlzIG5vdCBlc3RhYmxpc2hlZCBhcyBtaW5pLXN3ZS1hZ2VudC4","scoreUnit":"percent","url":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"swe-bench-multimodal","name":"SWE-bench Multimodal","preferredHarnessId":"swe-bench-multimodal:anthropic-fable-5-1-card:QW50aHJvcGljIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IGFkYXB0aXZlIHRoaW5raW5nIGF0IG1heCBlZmZvcnQsIGRlZmF1bHQgc2FtcGxpbmcsIGZpdmUtdHJpYWwgbWVhbiB1bmxlc3Mgc2VjdGlvbiBzcGVjaWZpZXMgb3RoZXJ3aXNlOyBjb250ZXh0IGF0IG1vc3QgMU0uIENvbXBhcmF0b3IgY29uZmlndXJhdGlvbiBmb2xsb3dzIGNpdGVkIHByaW9yIGNhcmQgb3IgYm9hcmQuIEFnZW50IGlkZW50aXR5IGlzIG5vdCBlc3RhYmxpc2hlZCBhcyBtaW5pLXN3ZS1hZ2VudC4","scoreUnit":"percent","url":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"deepswe-1-1-percent","name":"DeepSWE · 1.1","preferredHarnessId":"deepswe-1-1-percent:anthropic-fable-5-1-card:QW50aHJvcGljIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IGFkYXB0aXZlIHRoaW5raW5nIGF0IG1heCBlZmZvcnQsIGRlZmF1bHQgc2FtcGxpbmcsIGZpdmUtdHJpYWwgbWVhbiB1bmxlc3Mgc2VjdGlvbiBzcGVjaWZpZXMgb3RoZXJ3aXNlOyBjb250ZXh0IGF0IG1vc3QgMU0uIENvbXBhcmF0b3IgY29uZmlndXJhdGlvbiBmb2xsb3dzIGNpdGVkIHByaW9yIGNhcmQgb3IgYm9hcmQuIDExMyB0YXNrczsgb3JpZ2luYWwgaGlkZGVuLXRlc3QgZ3JhZGluZy4","scoreUnit":"percent","url":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","version":"1.1"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"frontiercode-1-1-extended","name":"FrontierCode · 1.1 Extended","preferredHarnessId":"frontiercode-1-1-extended:anthropic-fable-5-1-card:Q29nbml0aW9uIGFnZW50aWMgY29kaW5nOyBjb21wb3NpdGUgZnVuY3Rpb25hbCBhbmQgY29kZS1xdWFsaXR5IHNjb3JlLiBGYWJsZTUuMSBtZWRpdW0gZWZmb3J0OyBGYWJsZTUgeGhpZ2gu","scoreUnit":"percent","url":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","version":"1.1 Extended"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"frontiercode-1-1-main","name":"FrontierCode · 1.1 Main","preferredHarnessId":"frontiercode-1-1-main:anthropic-fable-5-1-card:Q29nbml0aW9uIGFnZW50aWMgY29kaW5nOyBjb21wb3NpdGUgZnVuY3Rpb25hbCBhbmQgY29kZS1xdWFsaXR5IHNjb3JlLiBGYWJsZTUuMSBtZWRpdW0gZWZmb3J0OyBGYWJsZTUgeGhpZ2gu","scoreUnit":"percent","url":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","version":"1.1 Main"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"frontierswe-2","name":"FrontierSWE · 2","preferredHarnessId":"frontierswe-2:anthropic-fable-5-1-card:UHJveGltYWwgYWdlbnQgaGFybmVzczsgbWF4IGVmZm9ydDsgMzQgdGFza3MsIGZpdmUgdHJpYWxzL3Rhc2s7IG1lYW4gc2NvcmUgb24gMC4uMSBzY2FsZS4","scoreUnit":"index","url":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","version":"2"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"terminal-bench-4-0-percent","name":"Terminal-Bench · 4.0","preferredHarnessId":"terminal-bench-4-0-percent:anthropic-fable-5-1-card:Q2xhdWRlIENvZGUgLS1iYXJlIG1heCBlZmZvcnQsIDE1IHRyaWFscy90YXNrIG92ZXIgNjYgdGFza3M7IEFudGhyb3BpYyBpbnRlcm5hbCByZXJ1bnMu","scoreUnit":"percent","url":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","version":"4.0"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"terminal-bench-science-0-1-percent","name":"Terminal-Bench-Science · 0.1","preferredHarnessId":"terminal-bench-science-0-1-percent:anthropic-fable-5-1-card:NzB0YXNrczsgQ2xhdWRlIENvZGUgLS1iYXJlIG1heDsgRmFibGUxMHRyaWFscy90YXNrLCBPcHVzMTI7IEdQVCBDb2RleCBDTEkgbWF4IGZyb20gcHVibGljIGJvYXJkLg","scoreUnit":"percent","url":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","version":"0.1"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"cursorbench-3-2-0","name":"CursorBench · 3.2.0","preferredHarnessId":"cursorbench-3-2-0:anthropic-fable-5-1-card:Q3Vyc29yIHByb2R1Y3Rpb24gYWdlbnQgaGFybmVzczsgaW5kZXBlbmRlbnRseSBtZWFzdXJlZCBieSBDdXJzb3I7IG1heCBlZmZvcnQu","scoreUnit":"percent","url":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","version":"3.2.0"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"programbench-166-golden-task-subset","name":"ProgramBench · 166 golden-task subset","preferredHarnessId":"programbench-166-golden-task-subset:anthropic-fable-5-1-card:bWluaS1zd2UtYWdlbnQgd2l0aG91dCBzaXgtaG91ciB0aW1lb3V0OyBleGNsdWRlczM0Zmxha3ktcmVmZXJlbmNlIHRhc2tzOyB0ZXN0cyByZXN0cmljdGVkIHRvIHJlZmVyZW5jZS1wYXNzaW5nIHRlc3RzOyB1cCB0bzFNY29udGV4dC4","scoreUnit":"percent","url":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","version":"166 golden-task subset"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"humanity-s-last-exam","name":"Humanity’s Last Exam","preferredHarnessId":"humanity-s-last-exam:anthropic-fable-5-1-card:RnVsbDI1MDBxdWVzdGlvbnM7IG5vIHRvb2xzOyBhdXRvIHRoaW5raW5nOzFNdG90YWwgdG9rZW4gY2FwOyBubyBjb21wYWN0aW9uOyBPcHVzNC42Z3JhZGVyOyByZXN0cmljdGVkIGZldGNoIGFuZCBjb250YW1pbmF0aW9uIHJldmlldyBmb3IgdG9vbHMu","scoreUnit":"percent","url":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"chartography","name":"Chartography","preferredHarnessId":"chartography:anthropic-fable-5-1-card:MTAwdGFza3M7IGFkYXB0aXZlIHRoaW5raW5nIG1heDsgZml2ZSBydW5zOyBubyB0b29sczsgdG9vbHMgY29uZGl0aW9uIGhhcyBjb250YWluZXIsIHN0YW5kYXJkIGxpYnJhcmllcyBhbmQgY3JvcCB0b29sLg","scoreUnit":"percent","url":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"benchcad-vision2code-1000-file-subset","name":"BenchCAD · Vision2Code 1000-file subset","preferredHarnessId":"benchcad-vision2code-1000-file-subset:anthropic-fable-5-1-card:UmFuZG9tMTAwMCBvZjE3OTAwZmlsZXM7IGZpdmUgcnVuczsgYWRhcHRpdmUgdGhpbmtpbmcgbWF4OyBubyB0b29sczsgY29ycmVjdGVkIGNhbWVyYSBwcm9tcHQsIHJhdyBzaGFwZXMgYWNjZXB0ZWQsIGxhc3QgY29kZSBmZW5jZSBwYXJzZWQu","scoreUnit":"index","url":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","version":"Vision2Code 1000-file subset"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"osworld-2-0-august2026-task-release","name":"OSWorld · 2.0 August2026 task release","preferredHarnessId":"osworld-2-0-august2026-task-release:anthropic-fable-5-1-card:cGFydGlhbCBwYXNzQDE7MTA4dGFza3M7IGZpdmUgcnVuczsxMDgwcDs1MDBzdGVwczsgbWF4IGVmZm9ydDsgT3B1czQuOGdyYWRlcjsgdGFzayBmaXhlczsgRmFibGUgc2FmZXR5IGludGVydmVudGlvbnMgc2NvcmUgemVyby4","scoreUnit":"percent","url":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","version":"2.0 August2026 task release"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"officeqa","name":"OfficeQA","preferredHarnessId":"officeqa:anthropic-fable-5-1-card:RXh0cmFjdGVkLXRleHQgVHJlYXN1cnkgY29ycHVzIGluIHNhbmRib3g7IGNvZGUgZXhlY3V0aW9uOyBwcm9kdWN0aW9uIE1lc3NhZ2VzIEFQSSB3aXRoIHNhZmVndWFyZHMvZmFsbGJhY2s7MTI4a291dHB1dCBjYXAuIFBybzEzM3F1ZXN0aW9ucy4","scoreUnit":"percent","url":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"officeqa-pro","name":"OfficeQA Pro","preferredHarnessId":"officeqa-pro:anthropic-fable-5-1-card:RXh0cmFjdGVkLXRleHQgVHJlYXN1cnkgY29ycHVzIGluIHNhbmRib3g7IGNvZGUgZXhlY3V0aW9uOyBwcm9kdWN0aW9uIE1lc3NhZ2VzIEFQSSB3aXRoIHNhZmVndWFyZHMvZmFsbGJhY2s7MTI4a291dHB1dCBjYXAuIFBybzEzM3F1ZXN0aW9ucy4","scoreUnit":"percent","url":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"legal-agent-benchmark-1235-task-public-subset","name":"Legal Agent Benchmark · 1235-task public subset","preferredHarnessId":"legal-agent-benchmark-1235-task-public-subset:anthropic-fable-5-1-card:YWxsLXBhc3M7IGZpdmUgcnVuczsgYWRhcHRpdmUgbWF4OyBpbnRlcm5hbCBiYXNoL1B5dGhvbiBoYXJuZXNzLCBTb25uZXQ0LjZqdWRnZTsxNmRlZmVjdGl2ZSB0YXNrcyBleGNsdWRlZDsgcHJvZHVjdGlvbiBzYWZlZ3VhcmRzL2ZhbGxiYWNrLg","scoreUnit":"percent","url":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","version":"1235-task public subset"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"legal-agent-benchmark-120-task-held-out-subset","name":"Legal Agent Benchmark · 120-task held-out subset","preferredHarnessId":"legal-agent-benchmark-120-task-held-out-subset:anthropic-fable-5-1-card:YWxsLXBhc3M7IEFydGlmaWNpYWwgQW5hbHlzaXMgaGFybmVzczsgeGhpZ2ggZWZmb3J0Lg","scoreUnit":"percent","url":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","version":"120-task held-out subset"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"gdpval-aa-2","name":"GDPval-AA · 2","preferredHarnessId":"gdpval-aa-2:anthropic-fable-5-1-card:QXJ0aWZpY2lhbCBBbmFseXNpcyBpbmRlcGVuZGVudCBhZ2VudGljIHNoZWxsL3dlYiBldmFsdWF0aW9uOzIyMEdEUHZhbGdvbGR0YXNrczsgYmxpbmQgcGFpcndpc2UgRWxvOyBtYXggZWZmb3J0IGZvciBDbGF1ZGUu","scoreUnit":"elo","url":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","version":"2"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"aa-briefcase","name":"AA-Briefcase","preferredHarnessId":"aa-briefcase:anthropic-fable-5-1-card:QXJ0aWZpY2lhbCBBbmFseXNpcyBsb25nLWhvcml6b24ga25vd2xlZGdlIHByb2plY3RzOyBydWJyaWMgYW5kIHBhbmVsIHBhaXJ3aXNlIGp1ZGdpbmc7IENsYXVkZSBtYXggZWZmb3J0Lg","scoreUnit":"elo","url":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"aa-briefcase-percent","name":"AA-Briefcase","preferredHarnessId":"aa-briefcase-percent:anthropic-fable-5-1-card:QXJ0aWZpY2lhbCBBbmFseXNpczsgbWF4IGVmZm9ydDsgY29tcG9uZW50IHJ1YnJpYyBwYXNzLg","scoreUnit":"percent","url":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"aa-briefcase-rating","name":"AA-Briefcase","preferredHarnessId":"aa-briefcase-rating:anthropic-fable-5-1-card:QXJ0aWZpY2lhbCBBbmFseXNpczsgbWF4IGVmZm9ydDsgY29tcG9uZW50IGFuYWx5dGljYWwgcXVhbGl0eS4","scoreUnit":"index","url":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"toolathlon-verified-june2026","name":"Toolathlon · Verified June2026","preferredHarnessId":"toolathlon-verified-june2026:anthropic-fable-5-1-card:UGFzc0AxOzEwOHRhc2tzL3RocmVlIHRyaWFsczsgaW50ZXJuYWwgaGFybmVzczsgbWF4IGVmZm9ydDsgRmFibGUgc2FmZWd1YXJkcytPcHVzNC44ZmFsbGJhY2s7IE9wdXMgc2FmZWd1YXJkcy9mYWxsYmFjayBkaXNhYmxlZDsgcGlubmVkIGNvbnRhaW5lcnMvZGF0YS4","scoreUnit":"percent","url":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","version":"Verified June2026"},{"combiner":{"inDefault":false},"higherIsBetter":false,"id":"toolathlon-verified-june2026-turns","name":"Toolathlon · Verified June2026","preferredHarnessId":"toolathlon-verified-june2026-turns:anthropic-fable-5-1-card:YXZlcmFnZSB0dXJuczsxMDh0YXNrcy90aHJlZSB0cmlhbHM7IGludGVybmFsIGhhcm5lc3M7IG1heCBlZmZvcnQ7IEZhYmxlIHNhZmVndWFyZHMrT3B1czQuOGZhbGxiYWNrOyBPcHVzIHNhZmVndWFyZHMvZmFsbGJhY2sgZGlzYWJsZWQ7IHBpbm5lZCBjb250YWluZXJzL2RhdGEu","scoreUnit":"index","url":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","version":"Verified June2026"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"automationbench","name":"AutomationBench","preferredHarnessId":"automationbench:anthropic-fable-5-1-card:UHJpdmF0ZSBoZWxkLW91dCBib2FyZDsgc2ltdWxhdGVkIGJ1c2luZXNzLXdvcmtmbG93IGFwcCBBUElzOyBhbGwgYXNzZXJ0aW9ucyBtdXN0IHBhc3M7IG1heCBlZmZvcnQgc3RhdGVkIGZvciBGYWJsZTUuMS9PcHVzNS4","scoreUnit":"percent","url":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"arc-agi-1-percent","name":"ARC-AGI · 1","preferredHarnessId":"arc-agi-1-percent:anthropic-fable-5-1-card:QVJDIFByaXplIHNlbWktcHJpdmF0ZSB2YWxpZGF0aW9uOyB2ZXJpZmllZCBGYWJsZTUuMSBtYXggZWZmb3J0OyBjb21wYXJhdG9yIGZpZ3VyZXMgYXMgcmVwb3J0ZWQgaW4gc3VtbWFyeS4","scoreUnit":"percent","url":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","version":"1"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"arc-agi-2-percent-higher","name":"ARC-AGI · 2","preferredHarnessId":"arc-agi-2-percent-higher:anthropic-fable-5-1-card:QVJDIFByaXplIHNlbWktcHJpdmF0ZSB2YWxpZGF0aW9uOyB2ZXJpZmllZCBGYWJsZTUuMSBtYXggZWZmb3J0OyBjb21wYXJhdG9yIGZpZ3VyZXMgYXMgcmVwb3J0ZWQgaW4gc3VtbWFyeS4","scoreUnit":"percent","url":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","version":"2"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"healthbench","name":"HealthBench","preferredHarnessId":"healthbench:anthropic-fable-5-1-card:UmF3IHJ1YnJpYyBzY29yZTsgYWRhcHRpdmUgbWF4OyBmaXZlIHRyaWFsczsgbm8gdG9vbHMvY3VzdG9tIHN5c3RlbSBwcm9tcHQ7IE9wdXM0LjhncmFkZXI7IEZhYmxlNS4xIHNhZmV0eSBmYWxsYmFjayB0b09wdXM1Lg","scoreUnit":"percent","url":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"healthbench-professional-percent","name":"HealthBench Professional","preferredHarnessId":"healthbench-professional-percent:anthropic-fable-5-1-card:UmF3IHJ1YnJpYyBzY29yZTsgYWRhcHRpdmUgbWF4OyBmaXZlIHRyaWFsczsgbm8gdG9vbHMvY3VzdG9tIHN5c3RlbSBwcm9tcHQ7IE9wdXM0LjhncmFkZXI7IEZhYmxlNS4xIHNhZmV0eSBmYWxsYmFjayB0b09wdXM1Lg","scoreUnit":"percent","url":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"gmmlu","name":"GMMLU","preferredHarnessId":"gmmlu:anthropic-fable-5-1-card:TWVhbiBhY2N1cmFjeTQybGFuZ3VhZ2VzOyBhZGFwdGl2ZSBtYXg7IG9uZSB0cmlhbDsgbm8gdG9vbHMvY3VzdG9tIHN5c3RlbSBwcm9tcHRzLg","scoreUnit":"percent","url":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"milu","name":"MILU","preferredHarnessId":"milu:anthropic-fable-5-1-card:TWVhbiBhY2N1cmFjeTExbGFuZ3VhZ2VzOyBhZGFwdGl2ZSBtYXg7IGZpdmUgdHJpYWxzOyBubyB0b29scy9jdXN0b20gc3lzdGVtIHByb21wdHMu","scoreUnit":"percent","url":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"biomysterybench-human-solvable","name":"BioMysteryBench · Human Solvable","preferredHarnessId":"biomysterybench-human-solvable:anthropic-fable-5-1-card:QW50aHJvcGljIGxpZmUtc2NpZW5jZXMgZXZhbHVhdGlvbjsgYmFzaC9lZGl0b3IvcGFja2FnZXMgZXhjZXB0IHByb3RlaW4gZGVzaWduIG5vIHRvb2xzOyBwcm90b2NvbHMgdXNlIGJhc2gvZWRpdG9yL3NlYXJjaDsgVW5kZXJzdGFuZGluZyByZXZpc2VkIG5ldHdvcmsgcmVzdHJpY3Rpb24u","scoreUnit":"percent","url":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","version":"Human Solvable"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"biomysterybench-human-difficult","name":"BioMysteryBench · Human Difficult","preferredHarnessId":"biomysterybench-human-difficult:anthropic-fable-5-1-card:QW50aHJvcGljIGxpZmUtc2NpZW5jZXMgZXZhbHVhdGlvbjsgYmFzaC9lZGl0b3IvcGFja2FnZXMgZXhjZXB0IHByb3RlaW4gZGVzaWduIG5vIHRvb2xzOyBwcm90b2NvbHMgdXNlIGJhc2gvZWRpdG9yL3NlYXJjaDsgVW5kZXJzdGFuZGluZyByZXZpc2VkIG5ldHdvcmsgcmVzdHJpY3Rpb24u","scoreUnit":"percent","url":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","version":"Human Difficult"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"spatialbench-verified","name":"SpatialBench · Verified","preferredHarnessId":"spatialbench-verified:anthropic-fable-5-1-card:QW50aHJvcGljIGxpZmUtc2NpZW5jZXMgZXZhbHVhdGlvbjsgYmFzaC9lZGl0b3IvcGFja2FnZXMgZXhjZXB0IHByb3RlaW4gZGVzaWduIG5vIHRvb2xzOyBwcm90b2NvbHMgdXNlIGJhc2gvZWRpdG9yL3NlYXJjaDsgVW5kZXJzdGFuZGluZyByZXZpc2VkIG5ldHdvcmsgcmVzdHJpY3Rpb24u","scoreUnit":"percent","url":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","version":"Verified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"singlecellbench","name":"SingleCellBench","preferredHarnessId":"singlecellbench:anthropic-fable-5-1-card:QW50aHJvcGljIGxpZmUtc2NpZW5jZXMgZXZhbHVhdGlvbjsgYmFzaC9lZGl0b3IvcGFja2FnZXMgZXhjZXB0IHByb3RlaW4gZGVzaWduIG5vIHRvb2xzOyBwcm90b2NvbHMgdXNlIGJhc2gvZWRpdG9yL3NlYXJjaDsgVW5kZXJzdGFuZGluZyByZXZpc2VkIG5ldHdvcmsgcmVzdHJpY3Rpb24u","scoreUnit":"percent","url":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"proteingym-hard","name":"ProteinGym · Hard","preferredHarnessId":"proteingym-hard:anthropic-fable-5-1-card:QW50aHJvcGljIGxpZmUtc2NpZW5jZXMgZXZhbHVhdGlvbjsgYmFzaC9lZGl0b3IvcGFja2FnZXMgZXhjZXB0IHByb3RlaW4gZGVzaWduIG5vIHRvb2xzOyBwcm90b2NvbHMgdXNlIGJhc2gvZWRpdG9yL3NlYXJjaDsgVW5kZXJzdGFuZGluZyByZXZpc2VkIG5ldHdvcmsgcmVzdHJpY3Rpb24u","scoreUnit":"index","url":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","version":"Hard"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"protein-design-sequence-generation-revised-grader","name":"Protein Design · Sequence Generation revised grader","preferredHarnessId":"protein-design-sequence-generation-revised-grader:anthropic-fable-5-1-card:QW50aHJvcGljIGxpZmUtc2NpZW5jZXMgZXZhbHVhdGlvbjsgYmFzaC9lZGl0b3IvcGFja2FnZXMgZXhjZXB0IHByb3RlaW4gZGVzaWduIG5vIHRvb2xzOyBwcm90b2NvbHMgdXNlIGJhc2gvZWRpdG9yL3NlYXJjaDsgVW5kZXJzdGFuZGluZyByZXZpc2VkIG5ldHdvcmsgcmVzdHJpY3Rpb24u","scoreUnit":"percent","url":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","version":"Sequence Generation revised grader"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"protein-design-library-ranking","name":"Protein Design · Library Ranking","preferredHarnessId":"protein-design-library-ranking:anthropic-fable-5-1-card:QW50aHJvcGljIGxpZmUtc2NpZW5jZXMgZXZhbHVhdGlvbjsgYmFzaC9lZGl0b3IvcGFja2FnZXMgZXhjZXB0IHByb3RlaW4gZGVzaWduIG5vIHRvb2xzOyBwcm90b2NvbHMgdXNlIGJhc2gvZWRpdG9yL3NlYXJjaDsgVW5kZXJzdGFuZGluZyByZXZpc2VkIG5ldHdvcmsgcmVzdHJpY3Rpb24u","scoreUnit":"percent","url":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","version":"Library Ranking"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"organic-chemistry-2-revised","name":"Organic Chemistry · 2 revised","preferredHarnessId":"organic-chemistry-2-revised:anthropic-fable-5-1-card:QW50aHJvcGljIGxpZmUtc2NpZW5jZXMgZXZhbHVhdGlvbjsgYmFzaC9lZGl0b3IvcGFja2FnZXMgZXhjZXB0IHByb3RlaW4gZGVzaWduIG5vIHRvb2xzOyBwcm90b2NvbHMgdXNlIGJhc2gvZWRpdG9yL3NlYXJjaDsgVW5kZXJzdGFuZGluZyByZXZpc2VkIG5ldHdvcmsgcmVzdHJpY3Rpb24u","scoreUnit":"percent","url":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","version":"2 revised"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"protocols-troubleshooting","name":"Protocols · Troubleshooting","preferredHarnessId":"protocols-troubleshooting:anthropic-fable-5-1-card:QW50aHJvcGljIGxpZmUtc2NpZW5jZXMgZXZhbHVhdGlvbjsgYmFzaC9lZGl0b3IvcGFja2FnZXMgZXhjZXB0IHByb3RlaW4gZGVzaWduIG5vIHRvb2xzOyBwcm90b2NvbHMgdXNlIGJhc2gvZWRpdG9yL3NlYXJjaDsgVW5kZXJzdGFuZGluZyByZXZpc2VkIG5ldHdvcmsgcmVzdHJpY3Rpb24u","scoreUnit":"percent","url":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","version":"Troubleshooting"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"protocols-understanding-network-restricted","name":"Protocols · Understanding network-restricted","preferredHarnessId":"protocols-understanding-network-restricted:anthropic-fable-5-1-card:QW50aHJvcGljIGxpZmUtc2NpZW5jZXMgZXZhbHVhdGlvbjsgYmFzaC9lZGl0b3IvcGFja2FnZXMgZXhjZXB0IHByb3RlaW4gZGVzaWduIG5vIHRvb2xzOyBwcm90b2NvbHMgdXNlIGJhc2gvZWRpdG9yL3NlYXJjaDsgVW5kZXJzdGFuZGluZyByZXZpc2VkIG5ldHdvcmsgcmVzdHJpY3Rpb24u","scoreUnit":"percent","url":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","version":"Understanding network-restricted"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"agents-last-exam-not-specified","name":"Agents' Last Exam · not specified","preferredHarnessId":"agents-last-exam-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","scoreUnit":"percent","url":"https://openai.com/index/gpt-6-astra/","version":"not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline","name":"OSWorld 2.0 (v2026.08.08, offline set, partial score) · 2.0 / v2026.08.08 offline","preferredHarnessId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IG9mZmxpbmUgc2V0OyBwYXJ0aWFsIGNyZWRpdDsgdjIwMjYuMDguMDg7IG9mZmljaWFsIHRhc2svZ3JhZGluZyBzZXR0aW5ncw","scoreUnit":"percent","url":"https://openai.com/index/gpt-6-astra/","version":"2.0 / v2026.08.08 offline"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"screenspot-pro-no-tools-not-specified","name":"ScreenSpot-Pro (no tools) · not specified","preferredHarnessId":"screenspot-pro-no-tools-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IG5vIHRvb2xz","scoreUnit":"percent","url":"https://openai.com/index/gpt-6-astra/","version":"not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"automationbench-not-specified","name":"AutomationBench · not specified","preferredHarnessId":"automationbench-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","scoreUnit":"percent","url":"https://openai.com/index/gpt-6-astra/","version":"not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"benchcad-not-specified","name":"BenchCAD · not specified","preferredHarnessId":"benchcad-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHRvb2xzIGVuYWJsZWQ","scoreUnit":"percent","url":"https://openai.com/index/gpt-6-astra/","version":"not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"browsecomp-not-specified","name":"BrowseComp · not specified","preferredHarnessId":"browsecomp-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","scoreUnit":"percent","url":"https://openai.com/index/gpt-6-astra/","version":"not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"openscore-string-quartets-1-omr-ned-not-specified","name":"OpenScore String Quartets (1 - OMR-NED) · not specified","preferredHarnessId":"openscore-string-quartets-1-omr-ned-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","scoreUnit":"index","url":"https://openai.com/index/gpt-6-astra/","version":"not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"internal-design-tasks-not-specified","name":"Internal Design Tasks · not specified","preferredHarnessId":"internal-design-tasks-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","scoreUnit":"percent","url":"https://openai.com/index/gpt-6-astra/","version":"not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"internal-data-science-tasks-not-specified","name":"Internal Data Science Tasks · not specified","preferredHarnessId":"internal-data-science-tasks-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","scoreUnit":"percent","url":"https://openai.com/index/gpt-6-astra/","version":"not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"artificial-analysis-intelligence-index-v4-1-1-4-1-1","name":"Artificial Analysis Intelligence Index v4.1.1 · v4.1.1","preferredHarnessId":"artificial-analysis-intelligence-index-v4-1-1-4-1-1:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","scoreUnit":"index","url":"https://openai.com/index/gpt-6-astra/","version":"v4.1.1"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"terminal-bench-4-0","name":"Terminal-Bench 4.0 · 4.0","preferredHarnessId":"terminal-bench-4-0:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","scoreUnit":"percent","url":"https://openai.com/index/gpt-6-astra/","version":"4.0"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"deepswe-v1-1-1-1-percent","name":"DeepSWE v1.1 · v1.1","preferredHarnessId":"deepswe-v1-1-1-1-percent:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","scoreUnit":"percent","url":"https://openai.com/index/gpt-6-astra/","version":"v1.1"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"frontiercode-1-1-extended-score-1-1","name":"FrontierCode 1.1 Extended (score) · 1.1","preferredHarnessId":"frontiercode-1-1-extended-score-1-1:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IENvZGV4LXN0eWxlIGRldmVsb3BlciBpbnN0cnVjdGlvbiBvbiB0ZXN0cywgcmV1c2UgYW5kIHJlcG9zaXRvcnkgY29udmVudGlvbnM","scoreUnit":"percent","url":"https://openai.com/index/gpt-6-astra/","version":"1.1"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"frontiercode-1-1-main-score-1-1","name":"FrontierCode 1.1 Main (score) · 1.1","preferredHarnessId":"frontiercode-1-1-main-score-1-1:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IENvZGV4LXN0eWxlIGRldmVsb3BlciBpbnN0cnVjdGlvbiBvbiB0ZXN0cywgcmV1c2UgYW5kIHJlcG9zaXRvcnkgY29udmVudGlvbnM","scoreUnit":"percent","url":"https://openai.com/index/gpt-6-astra/","version":"1.1"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"internal-database-migration-tasks-not-specified","name":"Internal Database Migration Tasks · not specified","preferredHarnessId":"internal-database-migration-tasks-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","scoreUnit":"percent","url":"https://openai.com/index/gpt-6-astra/","version":"not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"artificial-analysis-coding-agent-index-v1-4-1-4","name":"Artificial Analysis Coding Agent Index v1.4 · v1.4","preferredHarnessId":"artificial-analysis-coding-agent-index-v1-4-1-4:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","scoreUnit":"index","url":"https://openai.com/index/gpt-6-astra/","version":"v1.4"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"terminal-bench-science-0-1-not-specified","name":"Terminal-Bench Science 0.1 · not specified","preferredHarnessId":"terminal-bench-science-0-1-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","scoreUnit":"percent","url":"https://openai.com/index/gpt-6-astra/","version":"not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"frontiermath-tier-4-2","name":"FrontierMath Tier 4 (v2) · v2","preferredHarnessId":"frontiermath-tier-4-2:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","scoreUnit":"percent","url":"https://openai.com/index/gpt-6-astra/","version":"v2"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"gpqa-diamond-not-specified","name":"GPQA Diamond · not specified","preferredHarnessId":"gpqa-diamond-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","scoreUnit":"percent","url":"https://openai.com/index/gpt-6-astra/","version":"not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"humanity-s-last-exam-w-tools-not-specified","name":"Humanity's Last Exam (w/ tools) · not specified","preferredHarnessId":"humanity-s-last-exam-w-tools-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHRvb2xzIGVuYWJsZWQ","scoreUnit":"percent","url":"https://openai.com/index/gpt-6-astra/","version":"not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"genebench-pro-13","name":"GeneBench Pro · v13","preferredHarnessId":"genebench-pro-13:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","scoreUnit":"percent","url":"https://openai.com/index/gpt-6-astra/","version":"v13"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"medchembench-internal-not-specified","name":"MedChemBench (Internal) · not specified","preferredHarnessId":"medchembench-internal-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","scoreUnit":"percent","url":"https://openai.com/index/gpt-6-astra/","version":"not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"lifescibench-gold-v1","name":"LifeSciBench · Gold v1","preferredHarnessId":"lifescibench-gold-v1:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","scoreUnit":"percent","url":"https://openai.com/index/gpt-6-astra/","version":"Gold v1"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"healthbench-professional-length-adjusted-not-specified","name":"HealthBench Professional (length-adjusted) · not specified","preferredHarnessId":"healthbench-professional-length-adjusted-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IG9mZmljaWFsIHBhcGVyIHNjb3Jpbmc7IGxlbmd0aC1hZGp1c3RlZA","scoreUnit":"percent","url":"https://openai.com/index/gpt-6-astra/","version":"not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"exploitbench-not-specified","name":"ExploitBench · not specified","preferredHarnessId":"exploitbench-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHJlZHVjZWQgb3IgYWJzZW50IHByb2R1Y3Rpb24gc2FmZWd1YXJkcw","scoreUnit":"percent","url":"https://openai.com/index/gpt-6-astra/","version":"not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"exploitgym-not-specified","name":"ExploitGym · not specified","preferredHarnessId":"exploitgym-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHJlZHVjZWQgb3IgYWJzZW50IHByb2R1Y3Rpb24gc2FmZWd1YXJkczsgdjEgb2ZmbGluZSBlbnZpcm9ubWVudDsgbm8gcnVudGltZSBwYWNrYWdlIGluc3RhbGxhdGlvbjsgdG9rZW4gY2FwcGVkLCBubyB3YWxsLWNsb2NrIGNhcA","scoreUnit":"percent","url":"https://openai.com/index/gpt-6-astra/","version":"not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"exploitbench-june-aug-2026-june-aug2026","name":"ExploitBench (June-Aug 2026) · June-Aug2026","preferredHarnessId":"exploitbench-june-aug-2026-june-aug2026:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHJlZHVjZWQgb3IgYWJzZW50IHByb2R1Y3Rpb24gc2FmZWd1YXJkczsgMjAgdnVsbmVyYWJpbGl0aWVzIC8gMTMgQ2hyb21lIHJlbGVhc2Vz","scoreUnit":"percent","url":"https://openai.com/index/gpt-6-astra/","version":"June-Aug2026"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"sre-bench-262-binaries-19-programs","name":"SRE-Bench · 262 binaries / 19 programs","preferredHarnessId":"sre-bench-262-binaries-19-programs:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHJlZHVjZWQgb3IgYWJzZW50IHByb2R1Y3Rpb24gc2FmZWd1YXJkczsgcGFzc0AxOyBhbGwgc2l4IG9iamVjdGl2ZXMgcmVxdWlyZWQ","scoreUnit":"percent","url":"https://openai.com/index/gpt-6-astra/","version":"262 binaries / 19 programs"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"sec-bench-pro-may2026-revised-root-cause-grader","name":"SEC-Bench Pro · May2026 / revised root-cause grader","preferredHarnessId":"sec-bench-pro-may2026-revised-root-cause-grader:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHJlZHVjZWQgb3IgYWJzZW50IHByb2R1Y3Rpb24gc2FmZWd1YXJkczsgTWF5MjAyNiBKYXZhU2NyaXB0IHN1YnNldCwxODMgdnVsbmVyYWJpbGl0aWVzOyByZXZpc2VkIGFnZW50IHJvb3QtY2F1c2UgZ3JhZGVy","scoreUnit":"percent","url":"https://openai.com/index/gpt-6-astra/","version":"May2026 / revised root-cause grader"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"openai-mrcr-v2-8-needle-256k-512k-2","name":"OpenAI MRCR v2 8-needle 256K-512K · v2","preferredHarnessId":"openai-mrcr-v2-8-needle-256k-512k-2:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IDgtbmVlZGxlIDI1NkstNTEySw","scoreUnit":"percent","url":"https://openai.com/index/gpt-6-astra/","version":"v2"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"openai-mrcr-v2-8-needle-512k-1m-2","name":"OpenAI MRCR v2 8-needle 512K-1M · v2","preferredHarnessId":"openai-mrcr-v2-8-needle-512k-1m-2:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IDgtbmVlZGxlIDUxMkstMU0","scoreUnit":"percent","url":"https://openai.com/index/gpt-6-astra/","version":"v2"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"arc-agi-3","name":"ARC-AGI-3 · 3","preferredHarnessId":"arc-agi-3:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IFJlc3BvbnNlcyBBUEkgaGFybmVzcyB3aXRoIHR3byBkb2N1bWVudGVkIHNldHRpbmcgY2hhbmdlcw","scoreUnit":"percent","url":"https://openai.com/index/gpt-6-astra/","version":"3"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"arc-agi-2-percent","name":"ARC-AGI-2 · 2","preferredHarnessId":"arc-agi-2-percent:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","scoreUnit":"percent","url":"https://openai.com/index/gpt-6-astra/","version":"2"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"arc-agi-1","name":"ARC-AGI-1 · 1","preferredHarnessId":"arc-agi-1:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","scoreUnit":"percent","url":"https://openai.com/index/gpt-6-astra/","version":"1"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"gdpval-aa-v2-2","name":"GDPval-AA v2 · v2","preferredHarnessId":"gdpval-aa-v2-2:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","scoreUnit":"elo","url":"https://openai.com/index/gpt-5-6/","version":"v2"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"management-consulting-tasks-internal-not-specified","name":"Management Consulting Tasks (Internal) · not specified","preferredHarnessId":"management-consulting-tasks-internal-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","scoreUnit":"percent","url":"https://openai.com/index/gpt-5-6/","version":"not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"big-finance-bench-not-specified","name":"Big Finance Bench · not specified","preferredHarnessId":"big-finance-bench-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","scoreUnit":"percent","url":"https://openai.com/index/gpt-5-6/","version":"not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"artificial-analysis-intelligence-index-v4-1-4-1","name":"Artificial Analysis Intelligence Index v4.1 · v4.1","preferredHarnessId":"artificial-analysis-intelligence-index-v4-1-4-1:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","scoreUnit":"index","url":"https://openai.com/index/gpt-5-6/","version":"v4.1"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"artificial-analysis-coding-agent-index-v1-1-1-1","name":"Artificial Analysis Coding Agent Index v1.1 · v1.1","preferredHarnessId":"artificial-analysis-coding-agent-index-v1-1-1-1:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","scoreUnit":"index","url":"https://openai.com/index/gpt-5-6/","version":"v1.1"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"swe-bench-pro-not-specified","name":"SWE-Bench Pro · not specified","preferredHarnessId":"swe-bench-pro-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","scoreUnit":"percent","url":"https://openai.com/index/gpt-5-6/","version":"not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"terminal-bench-2-1-percent","name":"Terminal-Bench 2.1 · 2.1","preferredHarnessId":"terminal-bench-2-1-percent:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","scoreUnit":"percent","url":"https://openai.com/index/gpt-5-6/","version":"2.1"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"genebench-pro-not-specified","name":"GeneBench Pro · not specified","preferredHarnessId":"genebench-pro-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","scoreUnit":"percent","url":"https://openai.com/index/gpt-5-6/","version":"not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"lifescibench-not-specified","name":"LifeSciBench · not specified","preferredHarnessId":"lifescibench-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","scoreUnit":"percent","url":"https://openai.com/index/gpt-5-6/","version":"not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"healthbench-professional-not-specified","name":"HealthBench Professional · not specified","preferredHarnessId":"healthbench-professional-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IG9mZmljaWFsIHBhcGVyIHNjb3Jpbmc7IGxlbmd0aC1hZGp1c3RlZA","scoreUnit":"percent","url":"https://openai.com/index/gpt-5-6/","version":"not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"osworld-2-0-percent","name":"OSWorld 2.0 · 2.0","preferredHarnessId":"osworld-2-0-percent:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","scoreUnit":"percent","url":"https://openai.com/index/gpt-5-6/","version":"2.0"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"benchcad-python-tool-not-specified","name":"BenchCAD (python tool) · not specified","preferredHarnessId":"benchcad-python-tool-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IFB5dGhvbiB0b29sIGVuYWJsZWQ","scoreUnit":"percent","url":"https://openai.com/index/gpt-5-6/","version":"not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"capture-the-flag-challenges-not-specified","name":"Capture-the-Flag Challenges · not specified","preferredHarnessId":"capture-the-flag-challenges-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IHJlZHVjZWQgb3IgYWJzZW50IHByb2R1Y3Rpb24gc2FmZWd1YXJkcw","scoreUnit":"percent","url":"https://openai.com/index/gpt-5-6/","version":"not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"sec-bench-pro-may2026-public-grader","name":"SEC-Bench Pro · May2026 / public grader","preferredHarnessId":"sec-bench-pro-may2026-public-grader:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IHJlZHVjZWQgb3IgYWJzZW50IHByb2R1Y3Rpb24gc2FmZWd1YXJkczsgcHVibGljIGdyYWRlcjsgTWF5MjAyNiBKYXZhU2NyaXB0IHN1YnNldA","scoreUnit":"percent","url":"https://openai.com/index/gpt-5-6/","version":"May2026 / public grader"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"internal-research-debugging-evaluation-not-specified","name":"Internal Research Debugging Evaluation · not specified","preferredHarnessId":"internal-research-debugging-evaluation-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","scoreUnit":"percent","url":"https://openai.com/index/gpt-5-6/","version":"not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"kernelgen-1p-not-specified","name":"KernelGen 1P · not specified","preferredHarnessId":"kernelgen-1p-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","scoreUnit":"percent","url":"https://openai.com/index/gpt-5-6/","version":"not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"nanogpt-not-specified","name":"NanoGPT · not specified","preferredHarnessId":"nanogpt-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","scoreUnit":"percent","url":"https://openai.com/index/gpt-5-6/","version":"not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"posttrainbench-lite-not-specified","name":"PostTrainBench Lite · not specified","preferredHarnessId":"posttrainbench-lite-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","scoreUnit":"percent","url":"https://openai.com/index/gpt-5-6/","version":"not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"rsi-index-not-specified","name":"RSI Index · not specified","preferredHarnessId":"rsi-index-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","scoreUnit":"percent","url":"https://openai.com/index/gpt-5-6/","version":"not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"mmmu-pro-no-tools-not-specified","name":"MMMU Pro (no tools) · not specified","preferredHarnessId":"mmmu-pro-no-tools-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IG5vIHRvb2xz","scoreUnit":"percent","url":"https://openai.com/index/gpt-5-6/","version":"not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"mmmu-pro-with-tools-not-specified","name":"MMMU Pro (with tools) · not specified","preferredHarnessId":"mmmu-pro-with-tools-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IHRvb2xzIGVuYWJsZWQ","scoreUnit":"percent","url":"https://openai.com/index/gpt-5-6/","version":"not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"gdp-pdf-not-specified","name":"gdp.pdf · not specified","preferredHarnessId":"gdp-pdf-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","scoreUnit":"percent","url":"https://openai.com/index/gpt-5-6/","version":"not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"frontiermath-tier-1-3-2","name":"FrontierMath Tier 1-3 (v2) · v2","preferredHarnessId":"frontiermath-tier-1-3-2:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","scoreUnit":"percent","url":"https://openai.com/index/gpt-5-6/","version":"v2"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"toolathlon-not-specified","name":"Toolathlon · not specified","preferredHarnessId":"toolathlon-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","scoreUnit":"percent","url":"https://openai.com/index/gpt-5-6/","version":"not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"graphwalks-bfs-256k-f1-not-specified","name":"GraphWalks BFS 256k f1 · not specified","preferredHarnessId":"graphwalks-bfs-256k-f1-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IEJGUyAyNTZrIGYx","scoreUnit":"percent","url":"https://openai.com/index/gpt-5-6/","version":"not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"graphwalks-bfs-1mil-f1-not-specified","name":"GraphWalks BFS 1mil f1 · not specified","preferredHarnessId":"graphwalks-bfs-1mil-f1-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IEJGUyAxbWlsIGYx","scoreUnit":"percent","url":"https://openai.com/index/gpt-5-6/","version":"not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"healthbench-professional-not-specified-score-0-100","name":"HealthBench Professional · not specified","preferredHarnessId":"healthbench-professional-not-specified-score-0-100:openai-astra-system-card:bGVuZ3RoLWFkanVzdGVkOyBvZmZpY2lhbCBIZWFsdGhCZW5jaCBzY29yaW5n","scoreUnit":"index","url":"https://deploymentsafety.openai.com/gpt-6-astra","version":"not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"healthbench-not-specified","name":"HealthBench · not specified","preferredHarnessId":"healthbench-not-specified:openai-astra-system-card:bGVuZ3RoLWFkanVzdGVkOyBvZmZpY2lhbCBIZWFsdGhCZW5jaCBzY29yaW5n","scoreUnit":"index","url":"https://deploymentsafety.openai.com/gpt-6-astra","version":"not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"healthbench-hard-not-specified","name":"HealthBench Hard · not specified","preferredHarnessId":"healthbench-hard-not-specified:openai-astra-system-card:bGVuZ3RoLWFkanVzdGVkOyBvZmZpY2lhbCBIZWFsdGhCZW5jaCBzY29yaW5n","scoreUnit":"index","url":"https://deploymentsafety.openai.com/gpt-6-astra","version":"not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"healthbench-consensus-not-specified","name":"HealthBench Consensus · not specified","preferredHarnessId":"healthbench-consensus-not-specified:openai-astra-system-card:bGVuZ3RoLWFkanVzdGVkOyBvZmZpY2lhbCBIZWFsdGhCZW5jaCBzY29yaW5n","scoreUnit":"index","url":"https://deploymentsafety.openai.com/gpt-6-astra","version":"not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"internal-research-debugging-evaluation-41-research-bugs-6-alignment-auditing-tasks","name":"Internal Research Debugging Evaluation · 41 research bugs; 6 alignment-auditing tasks","preferredHarnessId":"internal-research-debugging-evaluation-41-research-bugs-6-alignment-auditing-tasks:openai-astra-system-card:U3lzdGVtLWNhcmQgcmVwb3J0ZWQgY29uZmlndXJhdGlvbjsgcmVhc29uaW5nIGVmZm9ydCB1bnNwZWNpZmllZA","scoreUnit":"percent","url":"https://deploymentsafety.openai.com/gpt-6-astra","version":"41 research bugs; 6 alignment-auditing tasks"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"sandbox-bench-september2026-internal","name":"Sandbox Bench · September2026 internal","preferredHarnessId":"sandbox-bench-september2026-internal:openai-astra-system-card:MjIgaXNvbGF0ZWQgQ1RGLXN0eWxlIHRhcmdldHM7IHByb3RlY3RlZC1mbGFnIHN1Y2Nlc3MgbWV0cmlj","scoreUnit":"percent","url":"https://deploymentsafety.openai.com/gpt-6-astra","version":"September2026 internal"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"no-cot-math-time-horizon-not-specified","name":"No-CoT math time horizon · not specified","preferredHarnessId":"no-cot-math-time-horizon-not-specified:openai-astra-system-card:VUsgQUlTSTsgc2luZ2xlIGZvcndhcmQgcGFzczsgbm8gY2hhaW4gb2YgdGhvdWdodA","scoreUnit":"index","url":"https://deploymentsafety.openai.com/gpt-6-astra","version":"not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"protocolqa-open-ended-108-questions","name":"ProtocolQA Open-Ended · 108 questions","preferredHarnessId":"protocolqa-open-ended-108-questions:openai-astra-system-card:b2JzZXJ2ZWQgcGVyZm9ybWFuY2U","scoreUnit":"percent","url":"https://deploymentsafety.openai.com/gpt-6-astra","version":"108 questions"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"tacit-knowledge-and-troubleshooting-60-questions","name":"Tacit Knowledge and Troubleshooting · 60 questions","preferredHarnessId":"tacit-knowledge-and-troubleshooting-60-questions:openai-astra-system-card:b2JzZXJ2ZWQgcGVyZm9ybWFuY2U","scoreUnit":"percent","url":"https://deploymentsafety.openai.com/gpt-6-astra","version":"60 questions"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"troubleshootingbench-156-questions-52-protocols","name":"TroubleshootingBench · 156 questions / 52 protocols","preferredHarnessId":"troubleshootingbench-156-questions-52-protocols:openai-astra-system-card:b2JzZXJ2ZWQgcGVyZm9ybWFuY2U","scoreUnit":"percent","url":"https://deploymentsafety.openai.com/gpt-6-astra","version":"156 questions / 52 protocols"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"shp2-protein-function-prediction-3-unpublished-assay-datasets","name":"SHP2 Protein Function Prediction · 3 unpublished assay datasets","preferredHarnessId":"shp2-protein-function-prediction-3-unpublished-assay-datasets:openai-astra-system-card:U3lzdGVtLWNhcmQgcmVwb3J0ZWQgY29uZmlndXJhdGlvbjsgcmVhc29uaW5nIGVmZm9ydCB1bnNwZWNpZmllZA","scoreUnit":"index","url":"https://deploymentsafety.openai.com/gpt-6-astra","version":"3 unpublished assay datasets"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"coronavirus-ace2-cell-entry-screen-not-specified","name":"Coronavirus-ACE2 Cell-Entry Screen · not specified","preferredHarnessId":"coronavirus-ace2-cell-entry-screen-not-specified:openai-astra-system-card:U3lzdGVtLWNhcmQgcmVwb3J0ZWQgY29uZmlndXJhdGlvbjsgcmVhc29uaW5nIGVmZm9ydCB1bnNwZWNpZmllZA","scoreUnit":"index","url":"https://deploymentsafety.openai.com/gpt-6-astra","version":"not specified"},{"combiner":{"inDefault":false},"higherIsBetter":false,"id":"phage-plasmid-co-evolution-not-specified","name":"Phage-plasmid Co-evolution · not specified","preferredHarnessId":"phage-plasmid-co-evolution-not-specified:openai-astra-system-card:U3lzdGVtLWNhcmQgcmVwb3J0ZWQgY29uZmlndXJhdGlvbjsgcmVhc29uaW5nIGVmZm9ydCB1bnNwZWNpZmllZA","scoreUnit":"index","url":"https://deploymentsafety.openai.com/gpt-6-astra","version":"not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"terminal-bench-science-0-1","name":"Terminal-Bench Science 0.1 · 0.1","preferredHarnessId":"terminal-bench-science-0-1:openai-astra-launch:TG93ZXItY29zdCBzZXR0aW5nOyBleGFjdCBlZmZvcnQgbm90IHNwZWNpZmllZA","scoreUnit":"percent","url":"https://openai.com/index/gpt-6-astra/","version":"0.1"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"terminal-bench-2-1","name":"Terminal Bench 2.1 · 2.1","preferredHarnessId":"terminal-bench-2-1:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gUXdlbiBDbGF1ZGUgQ29kZSBhdmdAMTAsNWggdGltZW91dCwxMzEwNzIgb3V0cHV0OyBjb21wYXJhdG9ycyBiZXN0IHB1Ymxpc2hlZCBhY3Jvc3MgaGFybmVzc2VzLg","scoreUnit":"percent","url":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B","version":"2.1"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"swe-bench-pro-source-release-snapshot-version-not-specified","name":"SWE-bench Pro · source release snapshot; version not specified","preferredHarnessId":"swe-bench-pro-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gUmVmaW5lZCB0YXNrIHNldCB3aXRoIHByb2JsZW1hdGljIHRhc2tzIGNvcnJlY3RlZDsgYWxsIGJhc2VsaW5lcyByZXJ1biwgQ2xhdWRlIENvZGUsdGVtcDEsdG9wX3AuOTUsMjU2SyBjb250ZXh0Lg","scoreUnit":"percent","url":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"deepswe-1-1-pct","name":"DeepSWE 1.1 · 1.1","preferredHarnessId":"deepswe-1-1-pct:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gQmVzdCBvZiBDbGF1ZGUgQ29kZSBhbmQgbWluaS1TV0UtYWdlbnQ7IFF3ZW4gYmVzdCBDbGF1ZGUgQ29kZTsgdGVtcDEsdG9wX3AuOTUsMjU2Sy4","scoreUnit":"percent","url":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B","version":"1.1"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"nl2repo-bench-source-release-snapshot-version-not-specified","name":"NL2Repo-Bench · source release snapshot; version not specified","preferredHarnessId":"nl2repo-bench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","scoreUnit":"percent","url":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"frontierswe-source-release-snapshot-version-not-specified","name":"FrontierSWE · source release snapshot; version not specified","preferredHarnessId":"frontierswe-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","scoreUnit":"percent","url":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"mls-bench-lite-source-release-snapshot-version-not-specified","name":"MLS-Bench-Lite · source release snapshot; version not specified","preferredHarnessId":"mls-bench-lite-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","scoreUnit":"percent","url":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"paperbench-source-release-snapshot-version-not-specified","name":"PaperBench · source release snapshot; version not specified","preferredHarnessId":"paperbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gQmFzaWNBZ2VudCBDb2RlLURldjsgT3B1czQuNiBqdWRnZSwzcnVucywxMmggZWFjaC4","scoreUnit":"percent","url":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"androidbench-source-release-snapshot-version-not-specified","name":"AndroidBench · source release snapshot; version not specified","preferredHarnessId":"androidbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","scoreUnit":"percent","url":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"qwenswebench-source-release-snapshot-version-not-specified","name":"QwenSWEBench · source release snapshot; version not specified","preferredHarnessId":"qwenswebench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gSW50ZXJuYWwgc29mdHdhcmUgZW5naW5lZXJpbmcsQ2xhdWRlQ29kZSxhdmdAMyw4aCwzMjc2OCBvdXRwdXQsdGVtcDEsMjU2Sy4","scoreUnit":"percent","url":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"qwenqoderbench-source-release-snapshot-version-not-specified","name":"QwenQoderBench · source release snapshot; version not specified","preferredHarnessId":"qwenqoderbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gSW50ZXJuYWwgUW9kZXIgdGFza3MsQ2xhdWRlQ29kZSxhdmdANSw2aCwzMjc2OCBvdXRwdXQsdGVtcDEsMjU2Sy4","scoreUnit":"percent","url":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"qwenreactbench-source-release-snapshot-version-not-specified","name":"QwenReactBench · source release snapshot; version not specified","preferredHarnessId":"qwenreactbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gSW50ZXJuYWwgYmlsaW5ndWFsIEVOL0NOIFJlYWN0IGJlbmNobWFyayw3Y2F0ZWdvcmllcyxDbGF1ZGVDb2RlLHJlbmRlcittdWx0aW1vZGFsanVkZ2UsQlQvRWxvLg","scoreUnit":"elo","url":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"qwensvgbench-source-release-snapshot-version-not-specified","name":"QwenSVGBench · source release snapshot; version not specified","preferredHarnessId":"qwensvgbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gSW50ZXJuYWwgYmlsaW5ndWFsIEVOL0NOIFNWRyBiZW5jaG1hcmsscmVuZGVyK211bHRpbW9kYWxqdWRnZSxCVC9FbG8u","scoreUnit":"elo","url":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"coworkbench-source-release-snapshot-version-not-specified","name":"CoWorkBench · source release snapshot; version not specified","preferredHarnessId":"coworkbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gSW50ZXJuYWwgcHJvZmVzc2lvbmFsIHdvcmsgdGFza3MgYWNyb3NzIHNjaWVuY2UsZmluYW5jZSxsYXcsbWVkaWNhbCxwcm9kdWN0aXZpdHku","scoreUnit":"percent","url":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"workspacebench-source-release-snapshot-version-not-specified","name":"WorkSpaceBench · source release snapshot; version not specified","preferredHarnessId":"workspacebench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","scoreUnit":"percent","url":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"jobbench-source-release-snapshot-version-not-specified","name":"JobBench · source release snapshot; version not specified","preferredHarnessId":"jobbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","scoreUnit":"percent","url":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"skillsbench-source-release-snapshot-version-not-specified","name":"SkillsBench · source release snapshot; version not specified","preferredHarnessId":"skillsbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gdjEuMSBwdWJsaWM4N3Rhc2tzLDNydW5zOyBBbnRocm9waWMgQ2xhdWRlQ29kZSxPcGVuQUkgQ29kZXgsUXdlbiBPcGVuQ29kZS4","scoreUnit":"percent","url":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"agents-last-exam-pass-score-source-release-snapshot-version-not-specified","name":"Agents' Last Exam (Pass / Score) · source release snapshot; version not specified","preferredHarnessId":"agents-last-exam-pass-score-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gUGFzcw","scoreUnit":"percent","url":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"automation-bench-pass-1-source-release-snapshot-version-not-specified","name":"Automation-Bench (Pass@1) · source release snapshot; version not specified","preferredHarnessId":"automation-bench-pass-1-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","scoreUnit":"percent","url":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"toolathlon-verified-pass-1-source-release-snapshot-version-not-specified","name":"Toolathlon Verified (Pass@1) · source release snapshot; version not specified","preferredHarnessId":"toolathlon-verified-pass-1-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","scoreUnit":"percent","url":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"widesearch-source-release-snapshot-version-not-specified","name":"WideSearch · source release snapshot; version not specified","preferredHarnessId":"widesearch-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gSXRlbS1GMSBvdmVyNHJ1bnM7IFF3ZW4tQWdlbnQgZm9yIFF3ZW4sQ2xhdWRlQ29kZSBmb3IgY29tcGFyYXRvcnMu","scoreUnit":"percent","url":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"hle-w-tools-source-release-snapshot-version-not-specified-pct","name":"HLE w/ tools · source release snapshot; version not specified","preferredHarnessId":"hle-w-tools-source-release-snapshot-version-not-specified-pct:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","scoreUnit":"percent","url":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"gpqa-diamond-source-release-snapshot-version-not-specified","name":"GPQA Diamond · source release snapshot; version not specified","preferredHarnessId":"gpqa-diamond-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","scoreUnit":"percent","url":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"hle-source-release-snapshot-version-not-specified","name":"HLE · source release snapshot; version not specified","preferredHarnessId":"hle-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","scoreUnit":"percent","url":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"ifbench-source-release-snapshot-version-not-specified","name":"IFBench · source release snapshot; version not specified","preferredHarnessId":"ifbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","scoreUnit":"percent","url":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"onemillion-bench-expert-score-source-release-snapshot-version-not-specified","name":"$OneMillion-Bench (expert score) · source release snapshot; version not specified","preferredHarnessId":"onemillion-bench-expert-score-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","scoreUnit":"percent","url":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"healthbench-source-release-snapshot-version-not-specified","name":"HealthBench · source release snapshot; version not specified","preferredHarnessId":"healthbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","scoreUnit":"percent","url":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"plawbench-source-release-snapshot-version-not-specified","name":"PLawBench · source release snapshot; version not specified","preferredHarnessId":"plawbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","scoreUnit":"percent","url":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"prbench-legal-source-release-snapshot-version-not-specified","name":"PRBench-Legal · source release snapshot; version not specified","preferredHarnessId":"prbench-legal-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","scoreUnit":"percent","url":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"prbench-finance-source-release-snapshot-version-not-specified","name":"PRBench-Finance · source release snapshot; version not specified","preferredHarnessId":"prbench-finance-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","scoreUnit":"percent","url":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"mrcr-v2-256k-8-needle-source-release-snapshot-version-not-specified","name":"MRCR v2 256K (8-needle) · source release snapshot; version not specified","preferredHarnessId":"mrcr-v2-256k-8-needle-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","scoreUnit":"percent","url":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"longbench-v2-source-release-snapshot-version-not-specified","name":"LongBench v2 · source release snapshot; version not specified","preferredHarnessId":"longbench-v2-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","scoreUnit":"percent","url":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"critpt-source-release-snapshot-version-not-specified","name":"CritPt · source release snapshot; version not specified","preferredHarnessId":"critpt-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"aa-lcr-source-release-snapshot-version-not-specified","name":"AA-LCR · source release snapshot; version not specified","preferredHarnessId":"aa-lcr-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"hle-full-source-release-snapshot-version-not-specified","name":"HLE-Full · source release snapshot; version not specified","preferredHarnessId":"hle-full-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aG91dCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"deepswe-source-release-snapshot-version-not-specified","name":"DeepSWE · source release snapshot; version not specified","preferredHarnessId":"deepswe-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"programbench-source-release-snapshot-version-not-specified","name":"ProgramBench · source release snapshot; version not specified","preferredHarnessId":"programbench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"terminal-bench-2-1-pct","name":"Terminal-Bench 2.1 · 2.1","preferredHarnessId":"terminal-bench-2-1-pct:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K3","version":"2.1"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"swe-marathon-source-release-snapshot-version-not-specified","name":"SWE-Marathon · source release snapshot; version not specified","preferredHarnessId":"swe-marathon-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"posttrainbench-source-release-snapshot-version-not-specified","name":"PostTrainBench · source release snapshot; version not specified","preferredHarnessId":"posttrainbench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"scicode-source-release-snapshot-version-not-specified","name":"SciCode · source release snapshot; version not specified","preferredHarnessId":"scicode-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"kimi-code-bench-2-0","name":"Kimi Code Bench 2.0 · 2.0","preferredHarnessId":"kimi-code-bench-2-0:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K3","version":"2.0"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"browsecomp-source-release-snapshot-version-not-specified","name":"BrowseComp · source release snapshot; version not specified","preferredHarnessId":"browsecomp-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"deepsearchqa-f1-source-release-snapshot-version-not-specified","name":"DeepSearchQA (F1) · source release snapshot; version not specified","preferredHarnessId":"deepsearchqa-f1-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"researchrubrics-source-release-snapshot-version-not-specified","name":"ResearchRubrics · source release snapshot; version not specified","preferredHarnessId":"researchrubrics-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"gdpval-aa-v2-elo-source-release-snapshot-version-not-specified","name":"GDPval-AA v2 (Elo) · source release snapshot; version not specified","preferredHarnessId":"gdpval-aa-v2-elo-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","scoreUnit":"elo","url":"https://huggingface.co/moonshotai/Kimi-K3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"toolathlon-verified-source-release-snapshot-version-not-specified-pct","name":"Toolathlon-Verified · source release snapshot; version not specified","preferredHarnessId":"toolathlon-verified-source-release-snapshot-version-not-specified-pct:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"mcpmark-verified-source-release-snapshot-version-not-specified","name":"MCPMark-Verified · source release snapshot; version not specified","preferredHarnessId":"mcpmark-verified-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"mcp-atlas-source-release-snapshot-version-not-specified","name":"MCP-Atlas · source release snapshot; version not specified","preferredHarnessId":"mcp-atlas-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"automationbench-source-release-snapshot-version-not-specified","name":"AutomationBench · source release snapshot; version not specified","preferredHarnessId":"automationbench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"aa-briefcase-elo-source-release-snapshot-version-not-specified","name":"AA-Briefcase (Elo) · source release snapshot; version not specified","preferredHarnessId":"aa-briefcase-elo-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","scoreUnit":"elo","url":"https://huggingface.co/moonshotai/Kimi-K3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"agents-last-exam-source-release-snapshot-version-not-specified","name":"Agents' Last Exam · source release snapshot; version not specified","preferredHarnessId":"agents-last-exam-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"apex-agents-source-release-snapshot-version-not-specified","name":"APEX-Agents · source release snapshot; version not specified","preferredHarnessId":"apex-agents-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"officeqa-pro-source-release-snapshot-version-not-specified","name":"OfficeQA Pro · source release snapshot; version not specified","preferredHarnessId":"officeqa-pro-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"spreadsheetbench-2-source-release-snapshot-version-not-specified","name":"SpreadsheetBench 2 · source release snapshot; version not specified","preferredHarnessId":"spreadsheetbench-2-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"osworld-verified-source-release-snapshot-version-not-specified-pct","name":"OSWorld-Verified · source release snapshot; version not specified","preferredHarnessId":"osworld-verified-source-release-snapshot-version-not-specified-pct:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"osworld-2-0","name":"OSWorld 2.0 · 2.0","preferredHarnessId":"osworld-2-0:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K3","version":"2.0"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"saas-bench-source-release-snapshot-version-not-specified","name":"SaaS-Bench · source release snapshot; version not specified","preferredHarnessId":"saas-bench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"tau3-banking-source-release-snapshot-version-not-specified-pct","name":"τ³-Banking · source release snapshot; version not specified","preferredHarnessId":"tau3-banking-source-release-snapshot-version-not-specified-pct:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"harvey-lab-aa-source-release-snapshot-version-not-specified","name":"Harvey Lab-AA · source release snapshot; version not specified","preferredHarnessId":"harvey-lab-aa-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"corpfin-v2-source-release-snapshot-version-not-specified","name":"CorpFin v2 · source release snapshot; version not specified","preferredHarnessId":"corpfin-v2-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"finance-agent-v2-source-release-snapshot-version-not-specified","name":"Finance Agent v2 · source release snapshot; version not specified","preferredHarnessId":"finance-agent-v2-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"legal-research-bench-source-release-snapshot-version-not-specified","name":"Legal Research Bench · source release snapshot; version not specified","preferredHarnessId":"legal-research-bench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"worldvqa-forceanswer-source-release-snapshot-version-not-specified","name":"WorldVQA ForceAnswer · source release snapshot; version not specified","preferredHarnessId":"worldvqa-forceanswer-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"omnidocbench-source-release-snapshot-version-not-specified","name":"OmniDocBench · source release snapshot; version not specified","preferredHarnessId":"omnidocbench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"perceptionbench-source-release-snapshot-version-not-specified","name":"PerceptionBench · source release snapshot; version not specified","preferredHarnessId":"perceptionbench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"video-mme-w-sub-source-release-snapshot-version-not-specified","name":"Video-MME (w. sub) · source release snapshot; version not specified","preferredHarnessId":"video-mme-w-sub-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"mmvu-source-release-snapshot-version-not-specified","name":"MMVU · source release snapshot; version not specified","preferredHarnessId":"mmvu-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"babyvision-w-python-source-release-snapshot-version-not-specified","name":"BabyVision w/ python · source release snapshot; version not specified","preferredHarnessId":"babyvision-w-python-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"mmmu-pro-source-release-snapshot-version-not-specified","name":"MMMU-Pro · source release snapshot; version not specified","preferredHarnessId":"mmmu-pro-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aG91dCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"charxiv-rq-source-release-snapshot-version-not-specified","name":"CharXiv (RQ) · source release snapshot; version not specified","preferredHarnessId":"charxiv-rq-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aG91dCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"mathvision-source-release-snapshot-version-not-specified","name":"MathVision · source release snapshot; version not specified","preferredHarnessId":"mathvision-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aG91dCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"zerobench-pass-5-source-release-snapshot-version-not-specified","name":"ZeroBench (pass@5) · source release snapshot; version not specified","preferredHarnessId":"zerobench-pass-5-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aG91dCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"gdpval-source-release-snapshot-version-not-specified","name":"GDPVal · source release snapshot; version not specified","preferredHarnessId":"gdpval-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","scoreUnit":"percent","url":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"swe-bench-verified-source-release-snapshot-version-not-specified","name":"SWE-Bench Verified · source release snapshot; version not specified","preferredHarnessId":"swe-bench-verified-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","scoreUnit":"percent","url":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"swe-bench-multilingual-source-release-snapshot-version-not-specified","name":"SWE-Bench Multilingual · source release snapshot; version not specified","preferredHarnessId":"swe-bench-multilingual-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","scoreUnit":"percent","url":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"profbench-search-source-release-snapshot-version-not-specified","name":"ProfBench (Search) · source release snapshot; version not specified","preferredHarnessId":"profbench-search-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","scoreUnit":"percent","url":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"pinchbench-source-release-snapshot-version-not-specified","name":"PinchBench · source release snapshot; version not specified","preferredHarnessId":"pinchbench-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","scoreUnit":"percent","url":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"taubench-v3-airline-source-release-snapshot-version-not-specified","name":"TauBench V3 Airline · source release snapshot; version not specified","preferredHarnessId":"taubench-v3-airline-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","scoreUnit":"percent","url":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"taubench-v3-retail-source-release-snapshot-version-not-specified","name":"TauBench V3 Retail · source release snapshot; version not specified","preferredHarnessId":"taubench-v3-retail-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","scoreUnit":"percent","url":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"taubench-v3-telecom-source-release-snapshot-version-not-specified","name":"TauBench V3 Telecom · source release snapshot; version not specified","preferredHarnessId":"taubench-v3-telecom-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","scoreUnit":"percent","url":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"taubench-v3-banking-source-release-snapshot-version-not-specified","name":"TauBench V3 Banking · source release snapshot; version not specified","preferredHarnessId":"taubench-v3-banking-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","scoreUnit":"percent","url":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"taubench-v3-average-source-release-snapshot-version-not-specified","name":"TauBench V3 Average · source release snapshot; version not specified","preferredHarnessId":"taubench-v3-average-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","scoreUnit":"percent","url":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"vals-ai-financial-agent-1-1-without-web-search-1-1","name":"Vals.ai Financial Agent 1.1 without web search · 1.1","preferredHarnessId":"vals-ai-financial-agent-1-1-without-web-search-1-1:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","scoreUnit":"percent","url":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","version":"1.1"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"vals-ai-financial-agent-1-1-with-web-search-1-1","name":"Vals.ai Financial Agent 1.1 with web search · 1.1","preferredHarnessId":"vals-ai-financial-agent-1-1-with-web-search-1-1:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","scoreUnit":"percent","url":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","version":"1.1"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"ioi-2025-source-release-snapshot-version-not-specified","name":"IOI 2025 · source release snapshot; version not specified","preferredHarnessId":"ioi-2025-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","scoreUnit":"index","url":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"livecodebench-v6-source-release-snapshot-version-not-specified","name":"LiveCodeBench (v6) · source release snapshot; version not specified","preferredHarnessId":"livecodebench-v6-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","scoreUnit":"percent","url":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"imoanswerbench-no-tools-source-release-snapshot-version-not-specified","name":"IMOAnswerBench (no tools) · source release snapshot; version not specified","preferredHarnessId":"imoanswerbench-no-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","scoreUnit":"percent","url":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"imoanswerbench-with-tools-source-release-snapshot-version-not-specified","name":"IMOAnswerBench (with tools) · source release snapshot; version not specified","preferredHarnessId":"imoanswerbench-with-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","scoreUnit":"percent","url":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"apex-shortlist-no-tools-source-release-snapshot-version-not-specified","name":"Apex-Shortlist (no tools) · source release snapshot; version not specified","preferredHarnessId":"apex-shortlist-no-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","scoreUnit":"percent","url":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"apex-shortlist-with-tools-source-release-snapshot-version-not-specified","name":"Apex-Shortlist (with tools) · source release snapshot; version not specified","preferredHarnessId":"apex-shortlist-with-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","scoreUnit":"percent","url":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"gpqa-no-tools-source-release-snapshot-version-not-specified","name":"GPQA (no tools) · source release snapshot; version not specified","preferredHarnessId":"gpqa-no-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","scoreUnit":"percent","url":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"scicode-subtask-source-release-snapshot-version-not-specified","name":"SciCode (subtask) · source release snapshot; version not specified","preferredHarnessId":"scicode-subtask-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","scoreUnit":"percent","url":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"hle-no-tools-source-release-snapshot-version-not-specified","name":"HLE (no tools) · source release snapshot; version not specified","preferredHarnessId":"hle-no-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","scoreUnit":"percent","url":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"hle-with-tools-source-release-snapshot-version-not-specified","name":"HLE (with tools) · source release snapshot; version not specified","preferredHarnessId":"hle-with-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","scoreUnit":"percent","url":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"critpt-no-tools-source-release-snapshot-version-not-specified","name":"CritPt (no tools) · source release snapshot; version not specified","preferredHarnessId":"critpt-no-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","scoreUnit":"percent","url":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"mmlu-pro-source-release-snapshot-version-not-specified-pct","name":"MMLU-Pro · source release snapshot; version not specified","preferredHarnessId":"mmlu-pro-source-release-snapshot-version-not-specified-pct:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","scoreUnit":"percent","url":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"omniscience-accuracy-source-release-snapshot-version-not-specified","name":"OmniScience Accuracy · source release snapshot; version not specified","preferredHarnessId":"omniscience-accuracy-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","scoreUnit":"percent","url":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"ifbench-prompt-loose-source-release-snapshot-version-not-specified","name":"IFBench (prompt loose) · source release snapshot; version not specified","preferredHarnessId":"ifbench-prompt-loose-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","scoreUnit":"percent","url":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"multi-challenge-source-release-snapshot-version-not-specified","name":"Multi-Challenge · source release snapshot; version not specified","preferredHarnessId":"multi-challenge-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","scoreUnit":"percent","url":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"ruler-1m-source-release-snapshot-version-not-specified","name":"RULER (1M) · source release snapshot; version not specified","preferredHarnessId":"ruler-1m-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","scoreUnit":"percent","url":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"longbench-v2-lte-1m-source-release-snapshot-version-not-specified","name":"Longbench v2 (≤ 1M) · source release snapshot; version not specified","preferredHarnessId":"longbench-v2-lte-1m-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","scoreUnit":"percent","url":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"mmlu-prox-avg-en-de-fr-es-it-ja-zh-hi-pt-ko-source-release-snapshot-version-not-specified","name":"MMLU-ProX (avg en/de/fr/es/it/ja/zh/hi/pt/ko) · source release snapshot; version not specified","preferredHarnessId":"mmlu-prox-avg-en-de-fr-es-it-ja-zh-hi-pt-ko-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","scoreUnit":"percent","url":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"wmt24-en-to-xx-source-release-snapshot-version-not-specified","name":"WMT24++ (en→xx) · source release snapshot; version not specified","preferredHarnessId":"wmt24-en-to-xx-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","scoreUnit":"index","url":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"hle-wo-w-tools-source-release-snapshot-version-not-specified","name":"HLE (wo / w tools) · source release snapshot; version not specified","preferredHarnessId":"hle-wo-w-tools-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQuIHdpdGhvdXQgdG9vbHM","scoreUnit":"percent","url":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"nl2repo-source-release-snapshot-version-not-specified","name":"NL2Repo · source release snapshot; version not specified","preferredHarnessId":"nl2repo-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","scoreUnit":"percent","url":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"cybergym-source-release-snapshot-version-not-specified-pct","name":"Cybergym · source release snapshot; version not specified","preferredHarnessId":"cybergym-source-release-snapshot-version-not-specified-pct:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","scoreUnit":"percent","url":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"automationbench-public-source-release-snapshot-version-not-specified","name":"AutomationBench (Public) · source release snapshot; version not specified","preferredHarnessId":"automationbench-public-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","scoreUnit":"percent","url":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"dsbench-fullstack-source-release-snapshot-version-not-specified","name":"DSBench-FullStack † · source release snapshot; version not specified","preferredHarnessId":"dsbench-fullstack-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","scoreUnit":"percent","url":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"dsbench-hard-source-release-snapshot-version-not-specified","name":"DSBench-Hard † · source release snapshot; version not specified","preferredHarnessId":"dsbench-hard-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","scoreUnit":"percent","url":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"terminal-bench-3-0","name":"Terminal Bench 3.0 · 3.0","preferredHarnessId":"terminal-bench-3-0:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","scoreUnit":"percent","url":"https://huggingface.co/zai-org/GLM-5.3","version":"3.0"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"deepswe-1-1","name":"DeepSWE (v1.1) · v1.1","preferredHarnessId":"deepswe-1-1:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","scoreUnit":"percent","url":"https://huggingface.co/zai-org/GLM-5.3","version":"v1.1"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"programbench-almost-solved-source-release-snapshot-version-not-specified","name":"ProgramBench (Almost Solved) · source release snapshot; version not specified","preferredHarnessId":"programbench-almost-solved-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","scoreUnit":"percent","url":"https://huggingface.co/zai-org/GLM-5.3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"swe-marathon-1-1","name":"SWE-Marathon (v1.1) · v1.1","preferredHarnessId":"swe-marathon-1-1:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","scoreUnit":"percent","url":"https://huggingface.co/zai-org/GLM-5.3","version":"v1.1"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"cybergym-source-release-snapshot-version-not-specified","name":"CyberGym · source release snapshot; version not specified","preferredHarnessId":"cybergym-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","scoreUnit":"percent","url":"https://huggingface.co/zai-org/GLM-5.3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"exploitgym-2h-6h-source-release-snapshot-version-not-specified","name":"ExploitGym (2h / 6h) · source release snapshot; version not specified","preferredHarnessId":"exploitgym-2h-6h-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4gMmggU2luZ2xlLXJ1bjg2OXRhc2tzOyBBUEkgdGltZSBub3JtYWxpemVkIGJ5IG1vZGVsIFRQUyBwbHVzIG92ZXJoZWFkOyByZXBvcnRlZCBudW1iZXIgc29sdmVkOyBkb21haW4gd2hpdGVsaXN0Lg","scoreUnit":"index","url":"https://huggingface.co/zai-org/GLM-5.3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"exploitbench-source-release-snapshot-version-not-specified","name":"ExploitBench · source release snapshot; version not specified","preferredHarnessId":"exploitbench-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","scoreUnit":"percent","url":"https://huggingface.co/zai-org/GLM-5.3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"toolathlon-verified-source-release-snapshot-version-not-specified","name":"Toolathlon Verified · source release snapshot; version not specified","preferredHarnessId":"toolathlon-verified-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","scoreUnit":"percent","url":"https://huggingface.co/zai-org/GLM-5.3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"automationbench-1-0-6","name":"AutomationBench (v1.0.6) · v1.0.6","preferredHarnessId":"automationbench-1-0-6:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","scoreUnit":"percent","url":"https://huggingface.co/zai-org/GLM-5.3","version":"v1.0.6"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"agents-last-exam-ale-cli-source-release-snapshot-version-not-specified","name":"Agents' Last Exam (ALE-CLI) · source release snapshot; version not specified","preferredHarnessId":"agents-last-exam-ale-cli-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","scoreUnit":"percent","url":"https://huggingface.co/zai-org/GLM-5.3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"hle-w-tools-source-release-snapshot-version-not-specified","name":"HLE w/ Tools · source release snapshot; version not specified","preferredHarnessId":"hle-w-tools-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","scoreUnit":"percent","url":"https://huggingface.co/zai-org/GLM-5.3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"gdpval-aa-v2-source-release-snapshot-version-not-specified","name":"GDPval-AA v2 · source release snapshot; version not specified","preferredHarnessId":"gdpval-aa-v2-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","scoreUnit":"elo","url":"https://huggingface.co/zai-org/GLM-5.3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"aa-intelligence-index-source-release-snapshot-version-not-specified","name":"AA Intelligence Index · source release snapshot; version not specified","preferredHarnessId":"aa-intelligence-index-source-release-snapshot-version-not-specified:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","scoreUnit":"index","url":"https://x.ai/news/grok-4-6","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"cursorbench-v3-2-3-2","name":"CursorBench v3.2 · v3.2","preferredHarnessId":"cursorbench-v3-2-3-2:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","scoreUnit":"percent","url":"https://x.ai/news/grok-4-6","version":"v3.2"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"deepswe-v1-1-1-1-pct","name":"DeepSWE v1.1 · v1.1","preferredHarnessId":"deepswe-v1-1-1-1-pct:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","scoreUnit":"percent","url":"https://x.ai/news/grok-4-6","version":"v1.1"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"frontiercode-v1-1-extended-1-1","name":"FrontierCode v1.1 Extended · v1.1","preferredHarnessId":"frontiercode-v1-1-extended-1-1:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","scoreUnit":"percent","url":"https://x.ai/news/grok-4-6","version":"v1.1"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"terminal-bench-v3-0-3-0","name":"Terminal-Bench v3.0 · v3.0","preferredHarnessId":"terminal-bench-v3-0-3-0:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","scoreUnit":"percent","url":"https://x.ai/news/grok-4-6","version":"v3.0"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"apex-swe-source-release-snapshot-version-not-specified","name":"APEX-SWE · source release snapshot; version not specified","preferredHarnessId":"apex-swe-source-release-snapshot-version-not-specified:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","scoreUnit":"percent","url":"https://x.ai/news/grok-4-6","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"harvey-lab-vals-source-release-snapshot-version-not-specified","name":"Harvey LAB (Vals) · source release snapshot; version not specified","preferredHarnessId":"harvey-lab-vals-source-release-snapshot-version-not-specified:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","scoreUnit":"percent","url":"https://x.ai/news/grok-4-6","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"harvey-legal-agent-benchmark-source-release-snapshot-version-not-specified","name":"Harvey Legal Agent Benchmark · source release snapshot; version not specified","preferredHarnessId":"harvey-legal-agent-benchmark-source-release-snapshot-version-not-specified:google:R29vZ2xlIGxhdW5jaCBjaGFydDsgYmVuY2htYXJrIG1ldGhvZG9sb2d5IGxpbmtlZCBvbiBwYWdlOyByZXBvcnRlZCBjb21wYXJhdG9yIHNldHRpbmdzIHZhcnku","scoreUnit":"percent","url":"https://deepmind.google/models/gemini/flash/","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"hle-verified-source-release-snapshot-version-not-specified","name":"HLE-Verified · source release snapshot; version not specified","preferredHarnessId":"hle-verified-source-release-snapshot-version-not-specified:google:R29vZ2xlIGxhdW5jaCBjaGFydDsgYmVuY2htYXJrIG1ldGhvZG9sb2d5IGxpbmtlZCBvbiBwYWdlOyByZXBvcnRlZCBjb21wYXJhdG9yIHNldHRpbmdzIHZhcnku","scoreUnit":"percent","url":"https://deepmind.google/models/gemini/flash/","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"aime-2025-source-release-snapshot-version-not-specified","name":"AIME 2025 · source release snapshot; version not specified","preferredHarnessId":"aime-2025-source-release-snapshot-version-not-specified:microsoft:TUFJIGF2ZyA0IHJ1bnMsIHRlbXBlcmF0dXJlIDEsIHRvcC1wIC45Nzsgc2ltcGxlIFJlQWN0IGJhc2gvc3RyaW5nLXJlcGxhY2U7IFRlcm1pbmFsLUJlbmNoIGlnbm9yZXMgcHJlZGVmaW5lZCB0aW1lb3V0cy4gQ29tcGFyYXRvciBjb25maWd1cmF0aW9ucyBmcm9tIGNpdGVkIHJlcG9ydHMu","scoreUnit":"percent","url":"https://microsoft.ai/pdf/mai-thinking-1.pdf","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"aime-2026-source-release-snapshot-version-not-specified","name":"AIME 2026 · source release snapshot; version not specified","preferredHarnessId":"aime-2026-source-release-snapshot-version-not-specified:microsoft:TUFJIGF2ZyA0IHJ1bnMsIHRlbXBlcmF0dXJlIDEsIHRvcC1wIC45Nzsgc2ltcGxlIFJlQWN0IGJhc2gvc3RyaW5nLXJlcGxhY2U7IFRlcm1pbmFsLUJlbmNoIGlnbm9yZXMgcHJlZGVmaW5lZCB0aW1lb3V0cy4gQ29tcGFyYXRvciBjb25maWd1cmF0aW9ucyBmcm9tIGNpdGVkIHJlcG9ydHMu","scoreUnit":"percent","url":"https://microsoft.ai/pdf/mai-thinking-1.pdf","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"hmmt-february-2026-source-release-snapshot-version-not-specified","name":"HMMT February 2026 · source release snapshot; version not specified","preferredHarnessId":"hmmt-february-2026-source-release-snapshot-version-not-specified:microsoft:TUFJIGF2ZyA0IHJ1bnMsIHRlbXBlcmF0dXJlIDEsIHRvcC1wIC45Nzsgc2ltcGxlIFJlQWN0IGJhc2gvc3RyaW5nLXJlcGxhY2U7IFRlcm1pbmFsLUJlbmNoIGlnbm9yZXMgcHJlZGVmaW5lZCB0aW1lb3V0cy4gQ29tcGFyYXRvciBjb25maWd1cmF0aW9ucyBmcm9tIGNpdGVkIHJlcG9ydHMu","scoreUnit":"percent","url":"https://microsoft.ai/pdf/mai-thinking-1.pdf","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"livecodebench-v6-source-release-snapshot-version-not-specified-pct","name":"LiveCodeBench v6 · source release snapshot; version not specified","preferredHarnessId":"livecodebench-v6-source-release-snapshot-version-not-specified-pct:microsoft:TUFJIGF2ZyA0IHJ1bnMsIHRlbXBlcmF0dXJlIDEsIHRvcC1wIC45Nzsgc2ltcGxlIFJlQWN0IGJhc2gvc3RyaW5nLXJlcGxhY2U7IFRlcm1pbmFsLUJlbmNoIGlnbm9yZXMgcHJlZGVmaW5lZCB0aW1lb3V0cy4gQ29tcGFyYXRvciBjb25maWd1cmF0aW9ucyBmcm9tIGNpdGVkIHJlcG9ydHMu","scoreUnit":"percent","url":"https://microsoft.ai/pdf/mai-thinking-1.pdf","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"terminal-bench-2-0","name":"Terminal-Bench 2.0 · 2.0","preferredHarnessId":"terminal-bench-2-0:microsoft:TUFJIGF2ZyA0IHJ1bnMsIHRlbXBlcmF0dXJlIDEsIHRvcC1wIC45Nzsgc2ltcGxlIFJlQWN0IGJhc2gvc3RyaW5nLXJlcGxhY2U7IFRlcm1pbmFsLUJlbmNoIGlnbm9yZXMgcHJlZGVmaW5lZCB0aW1lb3V0cy4gQ29tcGFyYXRvciBjb25maWd1cmF0aW9ucyBmcm9tIGNpdGVkIHJlcG9ydHMu","scoreUnit":"percent","url":"https://microsoft.ai/pdf/mai-thinking-1.pdf","version":"2.0"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"swe-bench-verified-source-release-snapshot-version-not-specified-pct","name":"SWE-bench Verified · source release snapshot; version not specified","preferredHarnessId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct:microsoft:TUFJIGF2ZyA0IHJ1bnMsIHRlbXBlcmF0dXJlIDEsIHRvcC1wIC45Nzsgc2ltcGxlIFJlQWN0IGJhc2gvc3RyaW5nLXJlcGxhY2U7IFRlcm1pbmFsLUJlbmNoIGlnbm9yZXMgcHJlZGVmaW5lZCB0aW1lb3V0cy4gQ29tcGFyYXRvciBjb25maWd1cmF0aW9ucyBmcm9tIGNpdGVkIHJlcG9ydHMu","scoreUnit":"percent","url":"https://microsoft.ai/pdf/mai-thinking-1.pdf","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"mmlu-pro-source-release-snapshot-version-not-specified","name":"MMLU Pro · source release snapshot; version not specified","preferredHarnessId":"mmlu-pro-source-release-snapshot-version-not-specified:microsoft:TWljcm9zb2Z0IG93biBldmFsdWF0aW9uIHN1aXRlOyBTb25uZXQgbWF4IHJlYXNvbmluZzsgVGFibGVzIDEyLzE5LiBMb25nQmVuY2hWMiBjYXBwZWQgMjU2ayw0MDggcXVlc3Rpb25zOyBDb3JwdXNRQSBHUFQtNS40IGhpZ2gganVkZ2U7IDQgcnVucy4","scoreUnit":"percent","url":"https://microsoft.ai/pdf/mai-thinking-1.pdf","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"simpleqa-verified-source-release-snapshot-version-not-specified","name":"SimpleQA Verified · source release snapshot; version not specified","preferredHarnessId":"simpleqa-verified-source-release-snapshot-version-not-specified:microsoft:TWljcm9zb2Z0IG93biBldmFsdWF0aW9uIHN1aXRlOyBTb25uZXQgbWF4IHJlYXNvbmluZzsgVGFibGVzIDEyLzE5LiBMb25nQmVuY2hWMiBjYXBwZWQgMjU2ayw0MDggcXVlc3Rpb25zOyBDb3JwdXNRQSBHUFQtNS40IGhpZ2gganVkZ2U7IDQgcnVucy4","scoreUnit":"percent","url":"https://microsoft.ai/pdf/mai-thinking-1.pdf","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"if-bench-source-release-snapshot-version-not-specified","name":"IF Bench · source release snapshot; version not specified","preferredHarnessId":"if-bench-source-release-snapshot-version-not-specified:microsoft:TWljcm9zb2Z0IG93biBldmFsdWF0aW9uIHN1aXRlOyBTb25uZXQgbWF4IHJlYXNvbmluZzsgVGFibGVzIDEyLzE5LiBMb25nQmVuY2hWMiBjYXBwZWQgMjU2ayw0MDggcXVlc3Rpb25zOyBDb3JwdXNRQSBHUFQtNS40IGhpZ2gganVkZ2U7IDQgcnVucy4","scoreUnit":"percent","url":"https://microsoft.ai/pdf/mai-thinking-1.pdf","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"advancedif-rubric-level-source-release-snapshot-version-not-specified","name":"AdvancedIF rubric-level · source release snapshot; version not specified","preferredHarnessId":"advancedif-rubric-level-source-release-snapshot-version-not-specified:microsoft:TWljcm9zb2Z0IG93biBldmFsdWF0aW9uIHN1aXRlOyBTb25uZXQgbWF4IHJlYXNvbmluZzsgVGFibGVzIDEyLzE5LiBMb25nQmVuY2hWMiBjYXBwZWQgMjU2ayw0MDggcXVlc3Rpb25zOyBDb3JwdXNRQSBHUFQtNS40IGhpZ2gganVkZ2U7IDQgcnVucy4","scoreUnit":"percent","url":"https://microsoft.ai/pdf/mai-thinking-1.pdf","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"multichallenge-source-release-snapshot-version-not-specified","name":"MultiChallenge · source release snapshot; version not specified","preferredHarnessId":"multichallenge-source-release-snapshot-version-not-specified:microsoft:TWljcm9zb2Z0IG93biBldmFsdWF0aW9uIHN1aXRlOyBTb25uZXQgbWF4IHJlYXNvbmluZzsgVGFibGVzIDEyLzE5LiBMb25nQmVuY2hWMiBjYXBwZWQgMjU2ayw0MDggcXVlc3Rpb25zOyBDb3JwdXNRQSBHUFQtNS40IGhpZ2gganVkZ2U7IDQgcnVucy4","scoreUnit":"percent","url":"https://microsoft.ai/pdf/mai-thinking-1.pdf","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"graphwalks-128k-source-release-snapshot-version-not-specified","name":"GraphWalks <=128k · source release snapshot; version not specified","preferredHarnessId":"graphwalks-128k-source-release-snapshot-version-not-specified:microsoft:TWljcm9zb2Z0IG93biBldmFsdWF0aW9uIHN1aXRlOyBTb25uZXQgbWF4IHJlYXNvbmluZzsgVGFibGVzIDEyLzE5LiBMb25nQmVuY2hWMiBjYXBwZWQgMjU2ayw0MDggcXVlc3Rpb25zOyBDb3JwdXNRQSBHUFQtNS40IGhpZ2gganVkZ2U7IDQgcnVucy4","scoreUnit":"percent","url":"https://microsoft.ai/pdf/mai-thinking-1.pdf","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"bfcl-v3-source-release-snapshot-version-not-specified","name":"BFCL v3 · source release snapshot; version not specified","preferredHarnessId":"bfcl-v3-source-release-snapshot-version-not-specified:microsoft:TWljcm9zb2Z0IG93biBldmFsdWF0aW9uIHN1aXRlOyBTb25uZXQgbWF4IHJlYXNvbmluZzsgVGFibGVzIDEyLzE5LiBMb25nQmVuY2hWMiBjYXBwZWQgMjU2ayw0MDggcXVlc3Rpb25zOyBDb3JwdXNRQSBHUFQtNS40IGhpZ2gganVkZ2U7IDQgcnVucy4","scoreUnit":"percent","url":"https://microsoft.ai/pdf/mai-thinking-1.pdf","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"healthbench-professional-source-release-snapshot-version-not-specified","name":"HealthBench Professional · source release snapshot; version not specified","preferredHarnessId":"healthbench-professional-source-release-snapshot-version-not-specified:microsoft:TWljcm9zb2Z0IG93biBldmFsdWF0aW9uIHN1aXRlOyBTb25uZXQgbWF4IHJlYXNvbmluZzsgVGFibGVzIDEyLzE5LiBMb25nQmVuY2hWMiBjYXBwZWQgMjU2ayw0MDggcXVlc3Rpb25zOyBDb3JwdXNRQSBHUFQtNS40IGhpZ2gganVkZ2U7IDQgcnVucy4","scoreUnit":"percent","url":"https://microsoft.ai/pdf/mai-thinking-1.pdf","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"medxpertqa-source-release-snapshot-version-not-specified","name":"MedXpertQA · source release snapshot; version not specified","preferredHarnessId":"medxpertqa-source-release-snapshot-version-not-specified:microsoft:TWljcm9zb2Z0IG93biBldmFsdWF0aW9uIHN1aXRlOyBTb25uZXQgbWF4IHJlYXNvbmluZzsgVGFibGVzIDEyLzE5LiBMb25nQmVuY2hWMiBjYXBwZWQgMjU2ayw0MDggcXVlc3Rpb25zOyBDb3JwdXNRQSBHUFQtNS40IGhpZ2gganVkZ2U7IDQgcnVucy4","scoreUnit":"percent","url":"https://microsoft.ai/pdf/mai-thinking-1.pdf","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"longbenchv2-source-release-snapshot-version-not-specified","name":"LongBenchV2 · source release snapshot; version not specified","preferredHarnessId":"longbenchv2-source-release-snapshot-version-not-specified:microsoft:TWljcm9zb2Z0IG93biBldmFsdWF0aW9uIHN1aXRlOyBTb25uZXQgbWF4IHJlYXNvbmluZzsgVGFibGVzIDEyLzE5LiBMb25nQmVuY2hWMiBjYXBwZWQgMjU2ayw0MDggcXVlc3Rpb25zOyBDb3JwdXNRQSBHUFQtNS40IGhpZ2gganVkZ2U7IDQgcnVucy4","scoreUnit":"percent","url":"https://microsoft.ai/pdf/mai-thinking-1.pdf","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"corpusqa-source-release-snapshot-version-not-specified","name":"CorpusQA · source release snapshot; version not specified","preferredHarnessId":"corpusqa-source-release-snapshot-version-not-specified:microsoft:TWljcm9zb2Z0IG93biBldmFsdWF0aW9uIHN1aXRlOyBTb25uZXQgbWF4IHJlYXNvbmluZzsgVGFibGVzIDEyLzE5LiBMb25nQmVuY2hWMiBjYXBwZWQgMjU2ayw0MDggcXVlc3Rpb25zOyBDb3JwdXNRQSBHUFQtNS40IGhpZ2gganVkZ2U7IDQgcnVucy4","scoreUnit":"percent","url":"https://microsoft.ai/pdf/mai-thinking-1.pdf","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"tau2-bench-telecom-source-release-snapshot-version-not-specified","name":"tau2-Bench Telecom · source release snapshot; version not specified","preferredHarnessId":"tau2-bench-telecom-source-release-snapshot-version-not-specified:cohere:Q29oZXJlIGxhdW5jaCBldmFsdWF0aW9uOyBOb3J0aCBtZXRyaWMgdXNlcyBpbnRlcm5hbCBMTE0ganVkZ2Uu","scoreUnit":"percent","url":"https://cohere.com/blog/command-a-plus","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"terminal-bench-hard-source-release-snapshot-version-not-specified","name":"Terminal-Bench Hard · source release snapshot; version not specified","preferredHarnessId":"terminal-bench-hard-source-release-snapshot-version-not-specified:cohere:Q29oZXJlIGxhdW5jaCBldmFsdWF0aW9uOyBOb3J0aCBtZXRyaWMgdXNlcyBpbnRlcm5hbCBMTE0ganVkZ2Uu","scoreUnit":"percent","url":"https://cohere.com/blog/command-a-plus","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"north-memory-usage-quality-source-release-snapshot-version-not-specified","name":"North Memory Usage Quality · source release snapshot; version not specified","preferredHarnessId":"north-memory-usage-quality-source-release-snapshot-version-not-specified:cohere:Q29oZXJlIGxhdW5jaCBldmFsdWF0aW9uOyBOb3J0aCBtZXRyaWMgdXNlcyBpbnRlcm5hbCBMTE0ganVkZ2Uu","scoreUnit":"percent","url":"https://cohere.com/blog/command-a-plus","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"mmmu-source-release-snapshot-version-not-specified","name":"MMMU · source release snapshot; version not specified","preferredHarnessId":"mmmu-source-release-snapshot-version-not-specified:cohere:Q29oZXJlIGxhdW5jaCBtdWx0aW1vZGFsIGV2YWx1YXRpb25zLg","scoreUnit":"percent","url":"https://cohere.com/blog/command-a-plus","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"mathvista-source-release-snapshot-version-not-specified","name":"MathVista · source release snapshot; version not specified","preferredHarnessId":"mathvista-source-release-snapshot-version-not-specified:cohere:Q29oZXJlIGxhdW5jaCBtdWx0aW1vZGFsIGV2YWx1YXRpb25zLg","scoreUnit":"percent","url":"https://cohere.com/blog/command-a-plus","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"charxiv-reasoning-source-release-snapshot-version-not-specified","name":"CharXiv Reasoning · source release snapshot; version not specified","preferredHarnessId":"charxiv-reasoning-source-release-snapshot-version-not-specified:cohere:Q29oZXJlIGxhdW5jaCBtdWx0aW1vZGFsIGV2YWx1YXRpb25zLg","scoreUnit":"percent","url":"https://cohere.com/blog/command-a-plus","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"aime-2025-avg-16-source-release-snapshot-version-not-specified","name":"AIME 2025 avg@16 · source release snapshot; version not specified","preferredHarnessId":"aime-2025-avg-16-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IFNvbm5ldDQuNiBleHRlcm5hbCBBUEkgdHJ1bmNhdGlvbiBjYXZlYXQu","scoreUnit":"percent","url":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"allenai-ifbench-source-release-snapshot-version-not-specified","name":"AllenAI IFBench · source release snapshot; version not specified","preferredHarnessId":"allenai-ifbench-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IFNvbm5ldDQuNiBleHRlcm5hbCBBUEkgdHJ1bmNhdGlvbiBjYXZlYXQu","scoreUnit":"percent","url":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"collie-source-release-snapshot-version-not-specified","name":"Collie · source release snapshot; version not specified","preferredHarnessId":"collie-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IFNvbm5ldDQuNiBleHRlcm5hbCBBUEkgdHJ1bmNhdGlvbiBjYXZlYXQu","scoreUnit":"percent","url":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"beyondaime-avg-16-source-release-snapshot-version-not-specified","name":"BeyondAIME avg@16 · source release snapshot; version not specified","preferredHarnessId":"beyondaime-avg-16-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IFNvbm5ldDQuNiBleHRlcm5hbCBBUEkgdHJ1bmNhdGlvbiBjYXZlYXQu","scoreUnit":"percent","url":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"tau3-telecom-source-release-snapshot-version-not-specified","name":"tau3 Telecom · source release snapshot; version not specified","preferredHarnessId":"tau3-telecom-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IHRhdTMgNCB0cmlhbHMsIEdQVDUuMiBsb3cgc2ltdWxhdG9yIGV4Y2VwdCBTaWVycmEgcmVwb3J0ZWQgY29tcGFyaXNvbnM7IEJyb3dzZUNvbXAgZGlzY2FyZC1hbGwgY29udGV4dCBhdDEwMGsu","scoreUnit":"percent","url":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"tau3-airline-source-release-snapshot-version-not-specified","name":"tau3 Airline · source release snapshot; version not specified","preferredHarnessId":"tau3-airline-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IHRhdTMgNCB0cmlhbHMsIEdQVDUuMiBsb3cgc2ltdWxhdG9yIGV4Y2VwdCBTaWVycmEgcmVwb3J0ZWQgY29tcGFyaXNvbnM7IEJyb3dzZUNvbXAgZGlzY2FyZC1hbGwgY29udGV4dCBhdDEwMGsu","scoreUnit":"percent","url":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"tau3-retail-source-release-snapshot-version-not-specified","name":"tau3 Retail · source release snapshot; version not specified","preferredHarnessId":"tau3-retail-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IHRhdTMgNCB0cmlhbHMsIEdQVDUuMiBsb3cgc2ltdWxhdG9yIGV4Y2VwdCBTaWVycmEgcmVwb3J0ZWQgY29tcGFyaXNvbnM7IEJyb3dzZUNvbXAgZGlzY2FyZC1hbGwgY29udGV4dCBhdDEwMGsu","scoreUnit":"percent","url":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"tau3-banking-source-release-snapshot-version-not-specified","name":"tau3 Banking · source release snapshot; version not specified","preferredHarnessId":"tau3-banking-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IHRhdTMgNCB0cmlhbHMsIEdQVDUuMiBsb3cgc2ltdWxhdG9yIGV4Y2VwdCBTaWVycmEgcmVwb3J0ZWQgY29tcGFyaXNvbnM7IEJyb3dzZUNvbXAgZGlzY2FyZC1hbGwgY29udGV4dCBhdDEwMGsu","scoreUnit":"percent","url":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"ifbench-prompt-loose-source-release-snapshot-version-not-specified-pct","name":"IFBench prompt loose · source release snapshot; version not specified","preferredHarnessId":"ifbench-prompt-loose-source-release-snapshot-version-not-specified-pct:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","scoreUnit":"percent","url":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"longcodebench-1m-source-release-snapshot-version-not-specified","name":"LongCodeBench 1M · source release snapshot; version not specified","preferredHarnessId":"longcodebench-1m-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","scoreUnit":"percent","url":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"tau2-telecom-source-release-snapshot-version-not-specified","name":"tau2 Telecom · source release snapshot; version not specified","preferredHarnessId":"tau2-telecom-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","scoreUnit":"percent","url":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"bfcl-v4-source-release-snapshot-version-not-specified","name":"BFCL v4 · source release snapshot; version not specified","preferredHarnessId":"bfcl-v4-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","scoreUnit":"percent","url":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"tau2-retail-verified-source-release-snapshot-version-not-specified","name":"tau2 Retail Verified · source release snapshot; version not specified","preferredHarnessId":"tau2-retail-verified-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","scoreUnit":"percent","url":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"tau2-airline-verified-source-release-snapshot-version-not-specified","name":"tau2 Airline Verified · source release snapshot; version not specified","preferredHarnessId":"tau2-airline-verified-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","scoreUnit":"percent","url":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"mcpatlas-source-release-snapshot-version-not-specified","name":"MCPAtlas · source release snapshot; version not specified","preferredHarnessId":"mcpatlas-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","scoreUnit":"percent","url":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"ocrbench-v2-average-accuracy-source-release-snapshot-version-not-specified","name":"OCRBench v2 average accuracy · source release snapshot; version not specified","preferredHarnessId":"ocrbench-v2-average-accuracy-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","scoreUnit":"percent","url":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"realkie-fcc-verified-alns-source-release-snapshot-version-not-specified","name":"RealKIE-FCC Verified ALNS · source release snapshot; version not specified","preferredHarnessId":"realkie-fcc-verified-alns-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","scoreUnit":"percent","url":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"qvhighlights-r1-0-5","name":"QVHighlights R1@0.5 · 0.5","preferredHarnessId":"qvhighlights-r1-0-5:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","scoreUnit":"percent","url":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","version":"0.5"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"screenspot-point-accuracy-source-release-snapshot-version-not-specified","name":"ScreenSpot point accuracy · source release snapshot; version not specified","preferredHarnessId":"screenspot-point-accuracy-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","scoreUnit":"percent","url":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"terminal-bench-1-0","name":"Terminal-Bench 1.0 · 1.0","preferredHarnessId":"terminal-bench-1-0:amazon:Tm92YTIgaW50ZXJuYWwgYWdlbnRpYyBzY2FmZm9sZDsgc2NhbGVkIGluZmVyZW5jZSBleHBsaWNpdGx5IHNlcGFyYXRlLiBDb21wYXJhdG9yIHNldHVwcyBmcm9tIGNpdGVkIHByb3ZpZGVyIHJlcG9ydC4","scoreUnit":"percent","url":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","version":"1.0"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"livecodebench-v5-july2024-january2025-source-release-snapshot-version-not-specified","name":"LiveCodeBench v5 July2024-January2025 · source release snapshot; version not specified","preferredHarnessId":"livecodebench-v5-july2024-january2025-source-release-snapshot-version-not-specified:amazon:Tm92YTIgaW50ZXJuYWwgYWdlbnRpYyBzY2FmZm9sZDsgc2NhbGVkIGluZmVyZW5jZSBleHBsaWNpdGx5IHNlcGFyYXRlLiBDb21wYXJhdG9yIHNldHVwcyBmcm9tIGNpdGVkIHByb3ZpZGVyIHJlcG9ydC4","scoreUnit":"percent","url":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"swe-bench-verified-scaled-inference-source-release-snapshot-version-not-specified","name":"SWE-bench Verified scaled inference · source release snapshot; version not specified","preferredHarnessId":"swe-bench-verified-scaled-inference-source-release-snapshot-version-not-specified:amazon:Tm92YTIgaW50ZXJuYWwgYWdlbnRpYyBzY2FmZm9sZDsgc2NhbGVkIGluZmVyZW5jZSBleHBsaWNpdGx5IHNlcGFyYXRlLiBDb21wYXJhdG9yIHNldHVwcyBmcm9tIGNpdGVkIHByb3ZpZGVyIHJlcG9ydC4","scoreUnit":"percent","url":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"sweatlas-qna-source-release-snapshot-version-not-specified","name":"SWEAtlas-QnA · source release snapshot; version not specified","preferredHarnessId":"sweatlas-qna-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","scoreUnit":"percent","url":"https://huggingface.co/MiniMaxAI/MiniMax-M3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"sweatlas-testwriting-source-release-snapshot-version-not-specified","name":"SWEAtlas-TestWriting · source release snapshot; version not specified","preferredHarnessId":"sweatlas-testwriting-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","scoreUnit":"percent","url":"https://huggingface.co/MiniMaxAI/MiniMax-M3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"swe-fficiency-source-release-snapshot-version-not-specified","name":"SWE-fficiency · source release snapshot; version not specified","preferredHarnessId":"swe-fficiency-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","scoreUnit":"percent","url":"https://huggingface.co/MiniMaxAI/MiniMax-M3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"livesqlbench-source-release-snapshot-version-not-specified","name":"LiveSQLBench · source release snapshot; version not specified","preferredHarnessId":"livesqlbench-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","scoreUnit":"percent","url":"https://huggingface.co/MiniMaxAI/MiniMax-M3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"cl-bench-source-release-snapshot-version-not-specified","name":"CL-bench · source release snapshot; version not specified","preferredHarnessId":"cl-bench-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","scoreUnit":"percent","url":"https://huggingface.co/MiniMaxAI/MiniMax-M3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"vibe-v2-source-release-snapshot-version-not-specified","name":"VIBE-V2 · source release snapshot; version not specified","preferredHarnessId":"vibe-v2-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","scoreUnit":"percent","url":"https://huggingface.co/MiniMaxAI/MiniMax-M3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"svg-bench-source-release-snapshot-version-not-specified","name":"SVG-Bench · source release snapshot; version not specified","preferredHarnessId":"svg-bench-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","scoreUnit":"percent","url":"https://huggingface.co/MiniMaxAI/MiniMax-M3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"kernelbench-hard-source-release-snapshot-version-not-specified","name":"KernelBench Hard · source release snapshot; version not specified","preferredHarnessId":"kernelbench-hard-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","scoreUnit":"percent","url":"https://huggingface.co/MiniMaxAI/MiniMax-M3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"draco-source-release-snapshot-version-not-specified","name":"DRACO · source release snapshot; version not specified","preferredHarnessId":"draco-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","scoreUnit":"percent","url":"https://huggingface.co/MiniMaxAI/MiniMax-M3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"gdpval-rubrics-source-release-snapshot-version-not-specified","name":"GDPval rubrics · source release snapshot; version not specified","preferredHarnessId":"gdpval-rubrics-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","scoreUnit":"percent","url":"https://huggingface.co/MiniMaxAI/MiniMax-M3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"bankertoolbench-source-release-snapshot-version-not-specified","name":"BankerToolBench · source release snapshot; version not specified","preferredHarnessId":"bankertoolbench-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","scoreUnit":"percent","url":"https://huggingface.co/MiniMaxAI/MiniMax-M3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"spreadsheetbench-v1-source-release-snapshot-version-not-specified","name":"SpreadsheetBench v1 · source release snapshot; version not specified","preferredHarnessId":"spreadsheetbench-v1-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","scoreUnit":"percent","url":"https://huggingface.co/MiniMaxAI/MiniMax-M3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"loca-bench-256k-source-release-snapshot-version-not-specified","name":"LOCA-Bench 256k · source release snapshot; version not specified","preferredHarnessId":"loca-bench-256k-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","scoreUnit":"percent","url":"https://huggingface.co/MiniMaxAI/MiniMax-M3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"claw-eval-source-release-snapshot-version-not-specified","name":"Claw-Eval · source release snapshot; version not specified","preferredHarnessId":"claw-eval-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","scoreUnit":"percent","url":"https://huggingface.co/MiniMaxAI/MiniMax-M3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"osworld-verified-source-release-snapshot-version-not-specified","name":"OSWorld Verified · source release snapshot; version not specified","preferredHarnessId":"osworld-verified-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","scoreUnit":"percent","url":"https://huggingface.co/MiniMaxAI/MiniMax-M3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"videommmu-source-release-snapshot-version-not-specified","name":"VideoMMMU · source release snapshot; version not specified","preferredHarnessId":"videommmu-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","scoreUnit":"percent","url":"https://huggingface.co/MiniMaxAI/MiniMax-M3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"videomme-with-subtitles-source-release-snapshot-version-not-specified","name":"VideoMME with subtitles · source release snapshot; version not specified","preferredHarnessId":"videomme-with-subtitles-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","scoreUnit":"percent","url":"https://huggingface.co/MiniMaxAI/MiniMax-M3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"yc-bench-source-release-snapshot-version-not-specified","name":"YC-Bench · source release snapshot; version not specified","preferredHarnessId":"yc-bench-source-release-snapshot-version-not-specified:minimax:TGF1bmNoIGZpZ3VyZTsgYWdlbnQgZmluYWwgYXNzZXRz","scoreUnit":"index","url":"https://huggingface.co/MiniMaxAI/MiniMax-M3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"imo-2025-source-release-snapshot-version-not-specified-points-42","name":"IMO 2025 · source release snapshot; version not specified","preferredHarnessId":"imo-2025-source-release-snapshot-version-not-specified-points-42:minimax:TWluaU1heCBwb2ludHMgb3V0IG9mNDI7IGNvbXBhcmF0b3IgcGVyY2VudGFnZXMgYXMgcHJpbnRlZA","scoreUnit":"index","url":"https://huggingface.co/MiniMaxAI/MiniMax-M3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"imo-2025-source-release-snapshot-version-not-specified","name":"IMO 2025 · source release snapshot; version not specified","preferredHarnessId":"imo-2025-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBwb2ludHMgb3V0IG9mNDI7IGNvbXBhcmF0b3IgcGVyY2VudGFnZXMgYXMgcHJpbnRlZA","scoreUnit":"percent","url":"https://huggingface.co/MiniMaxAI/MiniMax-M3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"usamo-2026-source-release-snapshot-version-not-specified-points-42","name":"USAMO 2026 · source release snapshot; version not specified","preferredHarnessId":"usamo-2026-source-release-snapshot-version-not-specified-points-42:minimax:TWluaU1heCBwb2ludHMgb3V0IG9mNDI7IGNvbXBhcmF0b3IgcGVyY2VudGFnZXMgYXMgcHJpbnRlZA","scoreUnit":"index","url":"https://huggingface.co/MiniMaxAI/MiniMax-M3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"usamo-2026-source-release-snapshot-version-not-specified","name":"USAMO 2026 · source release snapshot; version not specified","preferredHarnessId":"usamo-2026-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBwb2ludHMgb3V0IG9mNDI7IGNvbXBhcmF0b3IgcGVyY2VudGFnZXMgYXMgcHJpbnRlZA","scoreUnit":"percent","url":"https://huggingface.co/MiniMaxAI/MiniMax-M3","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"hle-with-tools-source-release-snapshot-version-not-specified-pct","name":"HLE with tools · source release snapshot; version not specified","preferredHarnessId":"hle-with-tools-source-release-snapshot-version-not-specified-pct:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","scoreUnit":"percent","url":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"hle-without-tools-source-release-snapshot-version-not-specified","name":"HLE without tools · source release snapshot; version not specified","preferredHarnessId":"hle-without-tools-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","scoreUnit":"percent","url":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"mrcr-v2-1m-8-needle-source-release-snapshot-version-not-specified","name":"MRCR v2 1M 8-needle · source release snapshot; version not specified","preferredHarnessId":"mrcr-v2-1m-8-needle-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","scoreUnit":"percent","url":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"osworld-2-0-binary-without-exec-2-0","name":"OSWorld 2.0 binary without exec · 2.0","preferredHarnessId":"osworld-2-0-binary-without-exec-2-0:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","scoreUnit":"percent","url":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","version":"2.0"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"osworld-2-0-partial-without-exec-2-0","name":"OSWorld 2.0 partial without exec · 2.0","preferredHarnessId":"osworld-2-0-partial-without-exec-2-0:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","scoreUnit":"percent","url":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","version":"2.0"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"webarena-verified-source-release-snapshot-version-not-specified","name":"WebArena Verified · source release snapshot; version not specified","preferredHarnessId":"webarena-verified-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","scoreUnit":"percent","url":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"deepsearchqa-source-release-snapshot-version-not-specified","name":"DeepSearchQA · source release snapshot; version not specified","preferredHarnessId":"deepsearchqa-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","scoreUnit":"percent","url":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"gdpval-aa-v2-elo-source-release-snapshot-version-not-specified-elo","name":"GDPval-AA v2 Elo · source release snapshot; version not specified","preferredHarnessId":"gdpval-aa-v2-elo-source-release-snapshot-version-not-specified-elo:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","scoreUnit":"elo","url":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"charxiv-reasoning-with-tools-source-release-snapshot-version-not-specified","name":"CharXiv Reasoning with tools · source release snapshot; version not specified","preferredHarnessId":"charxiv-reasoning-with-tools-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","scoreUnit":"percent","url":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"babyvision-with-tools-source-release-snapshot-version-not-specified","name":"BabyVision with tools · source release snapshot; version not specified","preferredHarnessId":"babyvision-with-tools-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","scoreUnit":"percent","url":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","version":"source release snapshot; version not specified"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"tau3-telecom-source-release-snapshot-version-as-labeled","name":"tau3 Telecom · source release snapshot; version as labeled","preferredHarnessId":"tau3-telecom-source-release-snapshot-version-as-labeled:mistral:TWF4aW11bSByZWFzb25pbmc7IE1pc3RyYWwgc2FtZS1sYWIgY29tcGFyaXNvbjsgdGF1MyA0dHJpYWxzIEdQVDUuMmxvdyB1c2VyIHNpbXVsYXRvcjsgQnJvd3NlQ29tcCBjb250ZXh0LWRpc2NhcmQgc2V0dXAgaW4gY2FyZC4","scoreUnit":"percent","url":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","version":"source release snapshot; version as labeled"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"tau3-airline-source-release-snapshot-version-as-labeled","name":"tau3 Airline · source release snapshot; version as labeled","preferredHarnessId":"tau3-airline-source-release-snapshot-version-as-labeled:mistral:TWF4aW11bSByZWFzb25pbmc7IE1pc3RyYWwgc2FtZS1sYWIgY29tcGFyaXNvbjsgdGF1MyA0dHJpYWxzIEdQVDUuMmxvdyB1c2VyIHNpbXVsYXRvcjsgQnJvd3NlQ29tcCBjb250ZXh0LWRpc2NhcmQgc2V0dXAgaW4gY2FyZC4","scoreUnit":"percent","url":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","version":"source release snapshot; version as labeled"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"tau3-retail-source-release-snapshot-version-as-labeled","name":"tau3 Retail · source release snapshot; version as labeled","preferredHarnessId":"tau3-retail-source-release-snapshot-version-as-labeled:mistral:TWF4aW11bSByZWFzb25pbmc7IE1pc3RyYWwgc2FtZS1sYWIgY29tcGFyaXNvbjsgdGF1MyA0dHJpYWxzIEdQVDUuMmxvdyB1c2VyIHNpbXVsYXRvcjsgQnJvd3NlQ29tcCBjb250ZXh0LWRpc2NhcmQgc2V0dXAgaW4gY2FyZC4","scoreUnit":"percent","url":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","version":"source release snapshot; version as labeled"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"tau3-banking-source-release-snapshot-version-as-labeled","name":"tau3 Banking · source release snapshot; version as labeled","preferredHarnessId":"tau3-banking-source-release-snapshot-version-as-labeled:mistral:TWF4aW11bSByZWFzb25pbmc7IE1pc3RyYWwgc2FtZS1sYWIgY29tcGFyaXNvbjsgdGF1MyA0dHJpYWxzIEdQVDUuMmxvdyB1c2VyIHNpbXVsYXRvcjsgQnJvd3NlQ29tcCBjb250ZXh0LWRpc2NhcmQgc2V0dXAgaW4gY2FyZC4","scoreUnit":"percent","url":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","version":"source release snapshot; version as labeled"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"browsecomp-source-release-snapshot-version-as-labeled","name":"BrowseComp · source release snapshot; version as labeled","preferredHarnessId":"browsecomp-source-release-snapshot-version-as-labeled:mistral:TWF4aW11bSByZWFzb25pbmc7IE1pc3RyYWwgc2FtZS1sYWIgY29tcGFyaXNvbjsgdGF1MyA0dHJpYWxzIEdQVDUuMmxvdyB1c2VyIHNpbXVsYXRvcjsgQnJvd3NlQ29tcCBjb250ZXh0LWRpc2NhcmQgc2V0dXAgaW4gY2FyZC4","scoreUnit":"percent","url":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","version":"source release snapshot; version as labeled"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"swe-bench-verified-source-release-snapshot-version-as-labeled","name":"SWE-bench Verified · source release snapshot; version as labeled","preferredHarnessId":"swe-bench-verified-source-release-snapshot-version-as-labeled:mistral:TWlzdHJhbCBwcmV2aW91cyBjb2RpbmcgbW9kZWwgY29tcGFyaXNvbiwgcHVibGlzaGVkIG1vZGVsLWNhcmQgaGFybmVzcyBzZXR0aW5ncy4","scoreUnit":"percent","url":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","version":"source release snapshot; version as labeled"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"ifbench-source-release-snapshot-version-as-labeled","name":"IFBench · source release snapshot; version as labeled","preferredHarnessId":"ifbench-source-release-snapshot-version-as-labeled:cohere:U2luZ2xlLXR1cm4gbG9vc2UscHJvbXB0IGFjY3VyYWN5LDI5NHByb21wdHMgeDUgcmVwZWF0cy4","scoreUnit":"percent","url":"https://cohere.com/blog/command-a-plus","version":"source release snapshot; version as labeled"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"aime-2025-source-release-snapshot-version-as-labeled","name":"AIME 2025 · source release snapshot; version as labeled","preferredHarnessId":"aime-2025-source-release-snapshot-version-as-labeled:cohere:T2ZmaWNpYWwzMHF1ZXN0aW9ucyB4MTAgcmVwZWF0czsgcGFzc0AxLg","scoreUnit":"percent","url":"https://cohere.com/blog/command-a-plus","version":"source release snapshot; version as labeled"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"scicode-source-release-snapshot-version-as-labeled","name":"SciCode · source release snapshot; version as labeled","preferredHarnessId":"scicode-source-release-snapshot-version-as-labeled:cohere:NjVwcm9ibGVtcy8yODhzdWJwcm9ibGVtczsgc2NpZW50aXN0LWFubm90YXRlZCBiYWNrZ3JvdW5kLg","scoreUnit":"percent","url":"https://cohere.com/blog/command-a-plus","version":"source release snapshot; version as labeled"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"north-agentic-question-answering-source-release-snapshot-version-as-labeled","name":"North Agentic Question Answering · source release snapshot; version as labeled","preferredHarnessId":"north-agentic-question-answering-source-release-snapshot-version-as-labeled:cohere:SW50ZXJuYWwgTm9ydGggZW50ZXJwcmlzZSBNQ1AgY2xvdWQtZmlsZSBRQSxMTE0ganVkZ2Uu","scoreUnit":"percent","url":"https://cohere.com/blog/command-a-plus","version":"source release snapshot; version as labeled"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"north-data-analysis-source-release-snapshot-version-as-labeled","name":"North Data Analysis · source release snapshot; version as labeled","preferredHarnessId":"north-data-analysis-source-release-snapshot-version-as-labeled:cohere:SW50ZXJuYWwgTm9ydGggdXBsb2FkZWQgc3ByZWFkc2hlZXQgZGF0YS1zY2llbmNlIHRhc2tzLExMTSBqdWRnZS4","scoreUnit":"percent","url":"https://cohere.com/blog/command-a-plus","version":"source release snapshot; version as labeled"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"mt-aime-2025-arabic-japanese-korean-source-release-snapshot-version-as-labeled","name":"MT-AIME 2025 Arabic/Japanese/Korean · source release snapshot; version as labeled","preferredHarnessId":"mt-aime-2025-arabic-japanese-korean-source-release-snapshot-version-as-labeled:cohere:SW50ZXJuYWwgQ29tbWFuZCBBIFRyYW5zbGF0ZSB0cmFuc2xhdGlvbnM7IEFyYWJpYyxKYXBhbmVzZSxLb3JlYW4u","scoreUnit":"percent","url":"https://cohere.com/blog/command-a-plus","version":"source release snapshot; version as labeled"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"wmt24-50-varieties-source-release-snapshot-version-as-labeled","name":"WMT24++ 50 varieties · source release snapshot; version as labeled","preferredHarnessId":"wmt24-50-varieties-source-release-snapshot-version-as-labeled:cohere:eENPTUVUeGwgYXZlcmFnZTUwIHZhcmlldGllcyxpbmNsdWRpbmcgaW50ZXJuYWwgSXJpc2gvTWFsdGVzZSB0cmFuc2xhdGlvbnMgYW5kIFNlcmJpYW4gdHJhbnNsaXRlcmF0aW9uLg","scoreUnit":"index","url":"https://cohere.com/blog/command-a-plus","version":"source release snapshot; version as labeled"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"charxiv-descriptive-source-release-snapshot-version-as-labeled","name":"CharXiv descriptive · source release snapshot; version as labeled","preferredHarnessId":"charxiv-descriptive-source-release-snapshot-version-as-labeled:cohere:U3RhbmRhcmQgbWV0aG9kb2xvZ3k7IGludGVnZXIgcm91bmRlZCBsYWJlbHMgaW4gY2hhcnQu","scoreUnit":"percent","url":"https://cohere.com/blog/command-a-plus","version":"source release snapshot; version as labeled"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"aa-intelligence-index-source-release-snapshot-version-as-labeled","name":"AA Intelligence Index · source release snapshot; version as labeled","preferredHarnessId":"aa-intelligence-index-source-release-snapshot-version-as-labeled:cohere:Q29oZXJlIHF1b3RlZCBBQSBsYXVuY2ggc25hcHNob3Q7IGluZGV4IHZlcnNpb24gdW5zcGVjaWZpZWQu","scoreUnit":"index","url":"https://cohere.com/blog/command-a-plus","version":"source release snapshot; version as labeled"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"jobbench","name":"JobBench","preferredHarnessId":"jobbench:meta-muse-1-3-report:NjV0YXNrczsgbWVhbiBydWJyaWMgc2NvcmU7IG9mZmljaWFsIE9wZW5Db2RlIGhhcm5lc3MgYW5kIGZpbGUtYXdhcmUgZ3JhZGVyOyByZWFzb25pbmcgbWF4","scoreUnit":"percent","url":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"osworld-partial-2-0-08-08","name":"OSWorld partial · 2.0 08.08","preferredHarnessId":"osworld-partial-2-0-08-08:meta-muse-1-3-report:MTA4dGFza3M7IGNvbW1vbiBpbnRlcm5hbCBHVUkgZnJhbWV3b3JrOyBleGVjdXRpb24tYmFzZWQgY2hlY2tlcnM7IHBhcnRpYWwgbWV0cmljOyByZWFzb25pbmcgbWF4","scoreUnit":"percent","url":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","version":"2.0 08.08"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"osworld-partial-2-0-06-24","name":"OSWorld partial · 2.0 06.24","preferredHarnessId":"osworld-partial-2-0-06-24:meta-muse-1-3-report:MTA4dGFza3M7IGNvbW1vbiBpbnRlcm5hbCBHVUkgZnJhbWV3b3JrOyBleGVjdXRpb24tYmFzZWQgY2hlY2tlcnM7IHBhcnRpYWwgbWV0cmljOyByZWFzb25pbmcgeGhpZ2g","scoreUnit":"percent","url":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","version":"2.0 06.24"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"osworld-binary-2-0-08-08","name":"OSWorld binary · 2.0 08.08","preferredHarnessId":"osworld-binary-2-0-08-08:meta-muse-1-3-report:MTA4dGFza3M7IGNvbW1vbiBpbnRlcm5hbCBHVUkgZnJhbWV3b3JrOyBleGVjdXRpb24tYmFzZWQgY2hlY2tlcnM7IGJpbmFyeSBtZXRyaWM7IHJlYXNvbmluZyBtYXg","scoreUnit":"percent","url":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","version":"2.0 08.08"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"osworld-binary-2-0-06-24","name":"OSWorld binary · 2.0 06.24","preferredHarnessId":"osworld-binary-2-0-06-24:meta-muse-1-3-report:MTA4dGFza3M7IGNvbW1vbiBpbnRlcm5hbCBHVUkgZnJhbWV3b3JrOyBleGVjdXRpb24tYmFzZWQgY2hlY2tlcnM7IGJpbmFyeSBtZXRyaWM7IHJlYXNvbmluZyB4aGlnaA","scoreUnit":"percent","url":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","version":"2.0 06.24"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"deepsearchqa","name":"DeepSearchQA","preferredHarnessId":"deepsearchqa:meta-muse-1-3-report:OTAwcXVlc3Rpb25zOyBjb21tb24gc2VhcmNoIGJhY2tlbmQvYnJvd3NlciBoYXJuZXNzOyBhbnN3ZXItc2V0IEYxOyByZWFzb25pbmcgbWF4","scoreUnit":"index","url":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"agentic-if-index-internal","name":"Agentic IF Index · internal","preferredHarnessId":"agentic-if-index-internal:meta-muse-1-3-report:SW50ZXJuYWwgY29tcG9zaXRlIGluc3RydWN0aW9uLWZvbGxvd2luZyBldmFsdWF0aW9uczsgbm8gZml4ZWQgdGFzayBjb3VudDsgcmVhc29uaW5nIG1heA","scoreUnit":"index","url":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","version":"internal"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"automationbench-public-v3","name":"AutomationBench · public v3","preferredHarnessId":"automationbench-public-v3:meta-muse-1-3-report:NjAwcHVibGljIHdvcmtmbG93IHRhc2tzOyBkZXRlcm1pbmlzdGljIGVuZC1zdGF0ZSBhc3NlcnRpb25zO3Bhc3NAMTsgcmVhc29uaW5nIG1heA","scoreUnit":"percent","url":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","version":"public v3"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"mrcr-2-256k-512k","name":"MRCR · v2 256K-512K","preferredHarnessId":"mrcr-2-256k-512k:meta-muse-1-3-report:OG5lZWRsZTsxMDBleGFtcGxlcy9iYW5kIHJlYmlubmVkIGJ5bzIwMGtfYmFzZTsgbm8gdG9vbHM7IHNlcXVlbmNlLW1hdGNoZXIgcmF0aW87IEdQVCBmcm9tT3BlbkFJY2FyZDsgcmVhc29uaW5nIG1heA","scoreUnit":"index","url":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","version":"v2 256K-512K"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"mrcr-2-512k-1m","name":"MRCR · v2 512K-1M","preferredHarnessId":"mrcr-2-512k-1m:meta-muse-1-3-report:OG5lZWRsZTsxMDBleGFtcGxlcy9iYW5kIHJlYmlubmVkIGJ5bzIwMGtfYmFzZTsgbm8gdG9vbHM7IHNlcXVlbmNlLW1hdGNoZXIgcmF0aW87IEdQVCBmcm9tT3BlbkFJY2FyZDsgcmVhc29uaW5nIG1heA","scoreUnit":"index","url":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","version":"v2 512K-1M"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"swe-atlas-codebase-qna","name":"SWE-Atlas Codebase QnA","preferredHarnessId":"swe-atlas-codebase-qna:meta-muse-1-3-report:MTI0dGFza3MvMTFyZXBvczsgcHVibGljUW5BIG1pbmktc3dlLWFnZW50IHJ1YnJpYyBwYXNzQDE7IEdQVC9PcHVzIG93bmNhcmRzOyByZWFzb25pbmcgbWF4","scoreUnit":"percent","url":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"terminal-bench-2-1-percent-higher","name":"Terminal-Bench · 2.1","preferredHarnessId":"terminal-bench-2-1-percent-higher:meta-muse-1-3-report:ODl0YXNrczsgZWFjaCBtb2RlbCBuYXRpdmVjb2RpbmcgaGFybmVzcyBpbiBpbnRlcm5hbGZyYW1ld29yay9jbG91ZHNhbmRib3g7IG9mZmljaWFsdmVyaWZpZXI7IEdQVCBvd25jYXJkOyByZWFzb25pbmcgbWF4OyBuYXRpdmUgaGFybmVzcyBmYW1pbHk6IE1ldGEgKGV4YWN0IGhhcm5lc3MgcmV2aXNpb24gbm90IHNwZWNpZmllZCk","scoreUnit":"percent","url":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","version":"2.1"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"gdp-pdf","name":"GDP.PDF","preferredHarnessId":"gdp-pdf:google-gemini-3-8-card:QWxsLXBhc3MgcmF0ZTsgYWxsbW9kZWxzIHNlbGZjb21wdXRlZCBieUdvb2dsZQ","scoreUnit":"percent","url":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"charxiv-reasoning","name":"CharXiv Reasoning","preferredHarnessId":"charxiv-reasoning:google-gemini-3-8-card:Tm8gdG9vbHM7IEdlbWluaS9HUFQvT3B1cyBzZWxmY29tcHV0ZWQ7IFNvbm5ldCBzZWxmcmVwb3J0ZWQ","scoreUnit":"percent","url":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"lvbench-static","name":"LVBench · static","preferredHarnessId":"lvbench-static:google-gemini-3-8-card:Tm8gdG9vbHM7MTAyNGZyYW1lcyBHZW1pbmkvR1BULDMwMGZyYW1lcyBDbGF1ZGUgZHVlQVBJbGltaXQ7IG1vZGVsLXNwZWNpZmljIGZyYW1lIGJ1ZGdldDogMTAyNA","scoreUnit":"percent","url":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","version":"static"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"lvbench-agentic","name":"LVBench · agentic","preferredHarnessId":"lvbench-agentic:google-gemini-3-8-card:Q2FyZCBsYWJlbHMgYWdlbnRpYzsgbGlua2VkbWV0aG9kb2xvZ3kgZGVzY3JpYmVzIG9ubHkgbm8tdG9vbHMgc3RhdGljIHNldHVw","scoreUnit":"percent","url":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","version":"agentic"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"osworld-partial-2-0-pre-08-08","name":"OSWorld partial · 2.0 pre-08.08","preferredHarnessId":"osworld-partial-2-0-pre-08-08:google-gemini-3-8-card:UGFydGlhbHNjb3JlOyBiYXRjaHRvb2xzOzEwODBwLzUwMHN0ZXBzOyBHZW1pbmkvU29ubmV0IGJlc3RvZjNydW5zOyBzY3JlZW5zaG90b25seTsgb2ZmaWNpYWxDVUFoYXJuZXNz","scoreUnit":"percent","url":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","version":"2.0 pre-08.08"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"osworld-partial-2-0-august2026-fixed-tasks","name":"OSWorld partial · 2.0 August2026 fixed tasks","preferredHarnessId":"osworld-partial-2-0-august2026-fixed-tasks:google-gemini-3-8-card:UGFydGlhbHNjb3JlOyBiYXRjaHRvb2xzOzEwODBwLzUwMHN0ZXBzOyBHZW1pbmkvU29ubmV0IGJlc3RvZjNydW5zOyBzY3JlZW5zaG90b25seTsgb2ZmaWNpYWxDVUFoYXJuZXNz","scoreUnit":"percent","url":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","version":"2.0 August2026 fixed tasks"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"osworld-partial-2-0-task-revision-unverified-gpt-5-6-sol","name":"OSWorld partial · 2.0 task revision unverified gpt-5.6-sol","preferredHarnessId":"osworld-partial-2-0-task-revision-unverified-gpt-5-6-sol:google-gemini-3-8-card:UGFydGlhbHNjb3JlOyBiYXRjaHRvb2xzOzEwODBwLzUwMHN0ZXBzOyBHZW1pbmkvU29ubmV0IGJlc3RvZjNydW5zOyBzY3JlZW5zaG90b25seTsgb2ZmaWNpYWxDVUFoYXJuZXNz","scoreUnit":"percent","url":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","version":"2.0 task revision unverified gpt-5.6-sol"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"osworld-partial-2-0-task-revision-unverified-gpt-5-6-terra","name":"OSWorld partial · 2.0 task revision unverified gpt-5.6-terra","preferredHarnessId":"osworld-partial-2-0-task-revision-unverified-gpt-5-6-terra:google-gemini-3-8-card:UGFydGlhbHNjb3JlOyBiYXRjaHRvb2xzOzEwODBwLzUwMHN0ZXBzOyBHZW1pbmkvU29ubmV0IGJlc3RvZjNydW5zOyBzY3JlZW5zaG90b25seTsgb2ZmaWNpYWxDVUFoYXJuZXNz","scoreUnit":"percent","url":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","version":"2.0 task revision unverified gpt-5.6-terra"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"labbench-2","name":"LABBench · 2","preferredHarnessId":"labbench-2:google-gemini-3-8-card:U2VsZmNvbXB1dGVkOyBMaW51eHRlcm1pbmFsLGJpb2luZm90b29scyxQeXRob24sUixuZXR3b3JrOyBtYWNyb2F2ZXJhZ2UxMXN1YnRhc2tz","scoreUnit":"percent","url":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","version":"2"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"hle-full-w-tools","name":"HLE-Full (w/ tools)","preferredHarnessId":"hle-full-w-tools:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEhMRS1GdWxsICh3LyB0b29scykuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMu","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K2.6","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"browsecomp-percent","name":"BrowseComp","preferredHarnessId":"browsecomp-percent:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEJyb3dzZUNvbXAuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMu","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K2.6","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"browsecomp-agent-swarm","name":"BrowseComp (Agent Swarm)","preferredHarnessId":"browsecomp-agent-swarm:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEJyb3dzZUNvbXAgKEFnZW50IFN3YXJtKS4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K2.6","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"deepsearchqa-f1-score","name":"DeepSearchQA (f1-score)","preferredHarnessId":"deepsearchqa-f1-score:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIERlZXBTZWFyY2hRQSAoZjEtc2NvcmUpLiBUaGlua2luZyBtb2RlOyB0ZW1wZXJhdHVyZSAxLjAsIHRvcC1wIDEuMCwgY29udGV4dCAyNjIxNDQuIENvbXBhcmF0b3Igc2V0dGluZ3M6IEdQVC01LjQgeGhpZ2gsIE9wdXM0LjYgbWF4LCBHZW1pbmkzLjFQcm8gaGlnaC4gU2VlIGJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMuIE9ubHkgb3duIEsyLjYgYW5kIHN0YXJyZWQgcmUtZXZhbHVhdGlvbnMgY2FuIGZvcm0gbmV3IGNvbXBhcmlzb25zLg","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K2.6","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"deepsearchqa-accuracy","name":"DeepSearchQA (accuracy)","preferredHarnessId":"deepsearchqa-accuracy:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIERlZXBTZWFyY2hRQSAoYWNjdXJhY3kpLiBUaGlua2luZyBtb2RlOyB0ZW1wZXJhdHVyZSAxLjAsIHRvcC1wIDEuMCwgY29udGV4dCAyNjIxNDQuIENvbXBhcmF0b3Igc2V0dGluZ3M6IEdQVC01LjQgeGhpZ2gsIE9wdXM0LjYgbWF4LCBHZW1pbmkzLjFQcm8gaGlnaC4gU2VlIGJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMuIE9ubHkgb3duIEsyLjYgYW5kIHN0YXJyZWQgcmUtZXZhbHVhdGlvbnMgY2FuIGZvcm0gbmV3IGNvbXBhcmlzb25zLg","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K2.6","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"widesearch-item-f1","name":"WideSearch (item-f1)","preferredHarnessId":"widesearch-item-f1:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIFdpZGVTZWFyY2ggKGl0ZW0tZjEpLiBUaGlua2luZyBtb2RlOyB0ZW1wZXJhdHVyZSAxLjAsIHRvcC1wIDEuMCwgY29udGV4dCAyNjIxNDQuIENvbXBhcmF0b3Igc2V0dGluZ3M6IEdQVC01LjQgeGhpZ2gsIE9wdXM0LjYgbWF4LCBHZW1pbmkzLjFQcm8gaGlnaC4gU2VlIGJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMuIE9ubHkgb3duIEsyLjYgYW5kIHN0YXJyZWQgcmUtZXZhbHVhdGlvbnMgY2FuIGZvcm0gbmV3IGNvbXBhcmlzb25zLg","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K2.6","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"toolathlon","name":"Toolathlon","preferredHarnessId":"toolathlon:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIFRvb2xhdGhsb24uIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMu","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K2.6","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"mcpmark","name":"MCPMark","preferredHarnessId":"mcpmark:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIE1DUE1hcmsuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMu","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K2.6","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"claw-eval-pass-3-percent","name":"Claw Eval (pass^3)","preferredHarnessId":"claw-eval-pass-3-percent:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIENsYXcgRXZhbCAocGFzc14zKS4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K2.6","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"claw-eval-pass-3","name":"Claw Eval (pass@3)","preferredHarnessId":"claw-eval-pass-3:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIENsYXcgRXZhbCAocGFzc0AzKS4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K2.6","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"apex-agents","name":"APEX-Agents","preferredHarnessId":"apex-agents:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEFQRVgtQWdlbnRzLiBUaGlua2luZyBtb2RlOyB0ZW1wZXJhdHVyZSAxLjAsIHRvcC1wIDEuMCwgY29udGV4dCAyNjIxNDQuIENvbXBhcmF0b3Igc2V0dGluZ3M6IEdQVC01LjQgeGhpZ2gsIE9wdXM0LjYgbWF4LCBHZW1pbmkzLjFQcm8gaGlnaC4gU2VlIGJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMuIE9ubHkgb3duIEsyLjYgYW5kIHN0YXJyZWQgcmUtZXZhbHVhdGlvbnMgY2FuIGZvcm0gbmV3IGNvbXBhcmlzb25zLg","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K2.6","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"osworld-verified-percent","name":"OSWorld-Verified","preferredHarnessId":"osworld-verified-percent:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIE9TV29ybGQtVmVyaWZpZWQuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMu","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K2.6","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"terminal-bench-2-0-terminus-2","name":"Terminal-Bench 2.0 (Terminus-2)","preferredHarnessId":"terminal-bench-2-0-terminus-2:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIFRlcm1pbmFsLUJlbmNoIDIuMCAoVGVybWludXMtMikuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMuIENvZGluZzogYXZlcmFnZSBvZiB0ZW4gcnVuczsgVGVybWluYWwgdXNlcyBUZXJtaW51czIgcHJlc2VydmUtdGhpbmtpbmc7IFNXRSB1c2VzIGluLWhvdXNlIFNXRS1hZ2VudC4","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K2.6","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"swe-bench-pro-percent","name":"SWE-Bench Pro","preferredHarnessId":"swe-bench-pro-percent:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIFNXRS1CZW5jaCBQcm8uIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMuIENvZGluZzogYXZlcmFnZSBvZiB0ZW4gcnVuczsgVGVybWluYWwgdXNlcyBUZXJtaW51czIgcHJlc2VydmUtdGhpbmtpbmc7IFNXRSB1c2VzIGluLWhvdXNlIFNXRS1hZ2VudC4","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K2.6","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"swe-bench-multilingual","name":"SWE-Bench Multilingual","preferredHarnessId":"swe-bench-multilingual:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIFNXRS1CZW5jaCBNdWx0aWxpbmd1YWwuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMuIENvZGluZzogYXZlcmFnZSBvZiB0ZW4gcnVuczsgVGVybWluYWwgdXNlcyBUZXJtaW51czIgcHJlc2VydmUtdGhpbmtpbmc7IFNXRSB1c2VzIGluLWhvdXNlIFNXRS1hZ2VudC4","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K2.6","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"swe-bench-verified-percent","name":"SWE-Bench Verified","preferredHarnessId":"swe-bench-verified-percent:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIFNXRS1CZW5jaCBWZXJpZmllZC4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4gQ29kaW5nOiBhdmVyYWdlIG9mIHRlbiBydW5zOyBUZXJtaW5hbCB1c2VzIFRlcm1pbnVzMiBwcmVzZXJ2ZS10aGlua2luZzsgU1dFIHVzZXMgaW4taG91c2UgU1dFLWFnZW50Lg","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K2.6","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"scicode","name":"SciCode","preferredHarnessId":"scicode:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIFNjaUNvZGUuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMuIENvZGluZzogYXZlcmFnZSBvZiB0ZW4gcnVuczsgVGVybWluYWwgdXNlcyBUZXJtaW51czIgcHJlc2VydmUtdGhpbmtpbmc7IFNXRSB1c2VzIGluLWhvdXNlIFNXRS1hZ2VudC4","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K2.6","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"ojbench-python","name":"OJBench (python)","preferredHarnessId":"ojbench-python:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIE9KQmVuY2ggKHB5dGhvbikuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMuIENvZGluZzogYXZlcmFnZSBvZiB0ZW4gcnVuczsgVGVybWluYWwgdXNlcyBUZXJtaW51czIgcHJlc2VydmUtdGhpbmtpbmc7IFNXRSB1c2VzIGluLWhvdXNlIFNXRS1hZ2VudC4","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K2.6","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"livecodebench-v6","name":"LiveCodeBench (v6)","preferredHarnessId":"livecodebench-v6:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIExpdmVDb2RlQmVuY2ggKHY2KS4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4gQ29kaW5nOiBhdmVyYWdlIG9mIHRlbiBydW5zOyBUZXJtaW5hbCB1c2VzIFRlcm1pbnVzMiBwcmVzZXJ2ZS10aGlua2luZzsgU1dFIHVzZXMgaW4taG91c2UgU1dFLWFnZW50Lg","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K2.6","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"hle-full","name":"HLE-Full","preferredHarnessId":"hle-full:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEhMRS1GdWxsLiBUaGlua2luZyBtb2RlOyB0ZW1wZXJhdHVyZSAxLjAsIHRvcC1wIDEuMCwgY29udGV4dCAyNjIxNDQuIENvbXBhcmF0b3Igc2V0dGluZ3M6IEdQVC01LjQgeGhpZ2gsIE9wdXM0LjYgbWF4LCBHZW1pbmkzLjFQcm8gaGlnaC4gU2VlIGJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMuIE9ubHkgb3duIEsyLjYgYW5kIHN0YXJyZWQgcmUtZXZhbHVhdGlvbnMgY2FuIGZvcm0gbmV3IGNvbXBhcmlzb25zLiBSZWFzb25pbmc6IDk4MzA0IGdlbmVyYXRpb24gdG9rZW5zOyBITEUgZnVsbCBzZXQu","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K2.6","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"aime-2026","name":"AIME 2026","preferredHarnessId":"aime-2026:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEFJTUUgMjAyNi4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4gUmVhc29uaW5nOiA5ODMwNCBnZW5lcmF0aW9uIHRva2VuczsgSExFIGZ1bGwgc2V0Lg","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K2.6","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"hmmt-2026-feb","name":"HMMT 2026 (Feb)","preferredHarnessId":"hmmt-2026-feb:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEhNTVQgMjAyNiAoRmViKS4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4gUmVhc29uaW5nOiA5ODMwNCBnZW5lcmF0aW9uIHRva2VuczsgSExFIGZ1bGwgc2V0Lg","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K2.6","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"imo-answerbench","name":"IMO-AnswerBench","preferredHarnessId":"imo-answerbench:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIElNTy1BbnN3ZXJCZW5jaC4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4gUmVhc29uaW5nOiA5ODMwNCBnZW5lcmF0aW9uIHRva2VuczsgSExFIGZ1bGwgc2V0Lg","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K2.6","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"gpqa-diamond-percent","name":"GPQA-Diamond","preferredHarnessId":"gpqa-diamond-percent:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEdQUUEtRGlhbW9uZC4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4gUmVhc29uaW5nOiA5ODMwNCBnZW5lcmF0aW9uIHRva2VuczsgSExFIGZ1bGwgc2V0Lg","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K2.6","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"mmmu-pro","name":"MMMU-Pro","preferredHarnessId":"mmmu-pro:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIE1NTVUtUHJvLiBUaGlua2luZyBtb2RlOyB0ZW1wZXJhdHVyZSAxLjAsIHRvcC1wIDEuMCwgY29udGV4dCAyNjIxNDQuIENvbXBhcmF0b3Igc2V0dGluZ3M6IEdQVC01LjQgeGhpZ2gsIE9wdXM0LjYgbWF4LCBHZW1pbmkzLjFQcm8gaGlnaC4gU2VlIGJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMuIE9ubHkgb3duIEsyLjYgYW5kIHN0YXJyZWQgcmUtZXZhbHVhdGlvbnMgY2FuIGZvcm0gbmV3IGNvbXBhcmlzb25zLiBWaXNpb246IDk4MzA0IHRva2VucywgYXZlcmFnZSBvZiB0aHJlZSBydW5zOyBQeXRob24gY29uZGl0aW9uIHVzZXMgNjU1MzYgdG9rZW5zIHBlciBzdGVwLCBhdCBtb3N0NTBzdGVwcy4","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K2.6","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"mmmu-pro-w-python","name":"MMMU-Pro (w/ python)","preferredHarnessId":"mmmu-pro-w-python:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIE1NTVUtUHJvICh3LyBweXRob24pLiBUaGlua2luZyBtb2RlOyB0ZW1wZXJhdHVyZSAxLjAsIHRvcC1wIDEuMCwgY29udGV4dCAyNjIxNDQuIENvbXBhcmF0b3Igc2V0dGluZ3M6IEdQVC01LjQgeGhpZ2gsIE9wdXM0LjYgbWF4LCBHZW1pbmkzLjFQcm8gaGlnaC4gU2VlIGJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMuIE9ubHkgb3duIEsyLjYgYW5kIHN0YXJyZWQgcmUtZXZhbHVhdGlvbnMgY2FuIGZvcm0gbmV3IGNvbXBhcmlzb25zLiBWaXNpb246IDk4MzA0IHRva2VucywgYXZlcmFnZSBvZiB0aHJlZSBydW5zOyBQeXRob24gY29uZGl0aW9uIHVzZXMgNjU1MzYgdG9rZW5zIHBlciBzdGVwLCBhdCBtb3N0NTBzdGVwcy4","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K2.6","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"charxiv-rq","name":"CharXiv (RQ)","preferredHarnessId":"charxiv-rq:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIENoYXJYaXYgKFJRKS4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4gVmlzaW9uOiA5ODMwNCB0b2tlbnMsIGF2ZXJhZ2Ugb2YgdGhyZWUgcnVuczsgUHl0aG9uIGNvbmRpdGlvbiB1c2VzIDY1NTM2IHRva2VucyBwZXIgc3RlcCwgYXQgbW9zdDUwc3RlcHMu","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K2.6","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"charxiv-rq-w-python","name":"CharXiv (RQ) (w/ python)","preferredHarnessId":"charxiv-rq-w-python:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIENoYXJYaXYgKFJRKSAody8gcHl0aG9uKS4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4gVmlzaW9uOiA5ODMwNCB0b2tlbnMsIGF2ZXJhZ2Ugb2YgdGhyZWUgcnVuczsgUHl0aG9uIGNvbmRpdGlvbiB1c2VzIDY1NTM2IHRva2VucyBwZXIgc3RlcCwgYXQgbW9zdDUwc3RlcHMu","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K2.6","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"mathvision","name":"MathVision","preferredHarnessId":"mathvision:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIE1hdGhWaXNpb24uIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMuIFZpc2lvbjogOTgzMDQgdG9rZW5zLCBhdmVyYWdlIG9mIHRocmVlIHJ1bnM7IFB5dGhvbiBjb25kaXRpb24gdXNlcyA2NTUzNiB0b2tlbnMgcGVyIHN0ZXAsIGF0IG1vc3Q1MHN0ZXBzLg","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K2.6","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"mathvision-w-python","name":"MathVision (w/ python)","preferredHarnessId":"mathvision-w-python:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIE1hdGhWaXNpb24gKHcvIHB5dGhvbikuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMuIFZpc2lvbjogOTgzMDQgdG9rZW5zLCBhdmVyYWdlIG9mIHRocmVlIHJ1bnM7IFB5dGhvbiBjb25kaXRpb24gdXNlcyA2NTUzNiB0b2tlbnMgcGVyIHN0ZXAsIGF0IG1vc3Q1MHN0ZXBzLg","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K2.6","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"babyvision","name":"BabyVision","preferredHarnessId":"babyvision:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEJhYnlWaXNpb24uIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMuIFZpc2lvbjogOTgzMDQgdG9rZW5zLCBhdmVyYWdlIG9mIHRocmVlIHJ1bnM7IFB5dGhvbiBjb25kaXRpb24gdXNlcyA2NTUzNiB0b2tlbnMgcGVyIHN0ZXAsIGF0IG1vc3Q1MHN0ZXBzLg","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K2.6","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"babyvision-w-python","name":"BabyVision (w/ python)","preferredHarnessId":"babyvision-w-python:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEJhYnlWaXNpb24gKHcvIHB5dGhvbikuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMuIFZpc2lvbjogOTgzMDQgdG9rZW5zLCBhdmVyYWdlIG9mIHRocmVlIHJ1bnM7IFB5dGhvbiBjb25kaXRpb24gdXNlcyA2NTUzNiB0b2tlbnMgcGVyIHN0ZXAsIGF0IG1vc3Q1MHN0ZXBzLg","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K2.6","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"v-w-python","name":"V* (w/ python)","preferredHarnessId":"v-w-python:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIFYqICh3LyBweXRob24pLiBUaGlua2luZyBtb2RlOyB0ZW1wZXJhdHVyZSAxLjAsIHRvcC1wIDEuMCwgY29udGV4dCAyNjIxNDQuIENvbXBhcmF0b3Igc2V0dGluZ3M6IEdQVC01LjQgeGhpZ2gsIE9wdXM0LjYgbWF4LCBHZW1pbmkzLjFQcm8gaGlnaC4gU2VlIGJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMuIE9ubHkgb3duIEsyLjYgYW5kIHN0YXJyZWQgcmUtZXZhbHVhdGlvbnMgY2FuIGZvcm0gbmV3IGNvbXBhcmlzb25zLiBWaXNpb246IDk4MzA0IHRva2VucywgYXZlcmFnZSBvZiB0aHJlZSBydW5zOyBQeXRob24gY29uZGl0aW9uIHVzZXMgNjU1MzYgdG9rZW5zIHBlciBzdGVwLCBhdCBtb3N0NTBzdGVwcy4","scoreUnit":"percent","url":"https://huggingface.co/moonshotai/Kimi-K2.6","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"cursorbench-4-0","name":"CursorBench · 4.0","preferredHarnessId":"cursorbench-4-0:xai-grok47:R3JvayA0LjcgWEhpZ2ggKERlZXBTV0U6IEhpZ2gpOyBHcm9rIDQuNiBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IENsYXVkZSBGYWJsZSA1LjEgTWF4LiBMYXVuY2ggY29tcGFyaXNvbjsgcGVyLW1vZGVsIGFnZW50IGhhcm5lc3MsIGV2YWx1YXRpb24gYnVkZ2V0IGFuZCByZXJ1biBwcm92ZW5hbmNlIGFyZSBub3Qgc3BlY2lmaWVkLg","scoreUnit":"percent","url":"https://x.ai/news/grok-4-7","version":"4.0"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"deepswe-v1-1-1-1","name":"DeepSWE v1.1 · 1.1","preferredHarnessId":"deepswe-v1-1-1-1:xai-grok47:R3JvayA0LjcgWEhpZ2ggKERlZXBTV0U6IEhpZ2gpOyBHcm9rIDQuNiBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IENsYXVkZSBGYWJsZSA1LjEgTWF4LiBMYXVuY2ggY29tcGFyaXNvbjsgcGVyLW1vZGVsIGFnZW50IGhhcm5lc3MsIGV2YWx1YXRpb24gYnVkZ2V0IGFuZCByZXJ1biBwcm92ZW5hbmNlIGFyZSBub3Qgc3BlY2lmaWVkLg","scoreUnit":"percent","url":"https://x.ai/news/grok-4-7","version":"1.1"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"eebench","name":"EEBench","preferredHarnessId":"eebench:xai-grok47:R3JvayA0LjcgWEhpZ2ggKERlZXBTV0U6IEhpZ2gpOyBHcm9rIDQuNiBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IENsYXVkZSBGYWJsZSA1LjEgTWF4LiBMYXVuY2ggY29tcGFyaXNvbjsgcGVyLW1vZGVsIGFnZW50IGhhcm5lc3MsIGV2YWx1YXRpb24gYnVkZ2V0IGFuZCByZXJ1biBwcm92ZW5hbmNlIGFyZSBub3Qgc3BlY2lmaWVkLg","scoreUnit":"percent","url":"https://x.ai/news/grok-4-7","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"aa-briefcase-1-1","name":"AA Briefcase · 1.1","preferredHarnessId":"aa-briefcase-1-1:xai-grok47:R3JvayA0LjcgWEhpZ2ggKERlZXBTV0U6IEhpZ2gpOyBHcm9rIDQuNiBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IENsYXVkZSBGYWJsZSA1LjEgTWF4LiBMYXVuY2ggY29tcGFyaXNvbjsgcGVyLW1vZGVsIGFnZW50IGhhcm5lc3MsIGV2YWx1YXRpb24gYnVkZ2V0IGFuZCByZXJ1biBwcm92ZW5hbmNlIGFyZSBub3Qgc3BlY2lmaWVkLg","scoreUnit":"elo","url":"https://x.ai/news/grok-4-7","version":"1.1"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"terminal-bench-4-0-pct","name":"Terminal-Bench · 4.0","preferredHarnessId":"terminal-bench-4-0-pct:xai-grok47:R3JvayA0LjcgWEhpZ2ggKERlZXBTV0U6IEhpZ2gpOyBHcm9rIDQuNiBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IENsYXVkZSBGYWJsZSA1LjEgTWF4LiBMYXVuY2ggY29tcGFyaXNvbjsgcGVyLW1vZGVsIGFnZW50IGhhcm5lc3MsIGV2YWx1YXRpb24gYnVkZ2V0IGFuZCByZXJ1biBwcm92ZW5hbmNlIGFyZSBub3Qgc3BlY2lmaWVkLg","scoreUnit":"percent","url":"https://x.ai/news/grok-4-7","version":"4.0"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"harvey-legal-agent-benchmark","name":"Harvey Legal Agent Benchmark","preferredHarnessId":"harvey-legal-agent-benchmark:xai-grok47:R3JvayA0LjcgWEhpZ2ggKERlZXBTV0U6IEhpZ2gpOyBHcm9rIDQuNiBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IENsYXVkZSBGYWJsZSA1LjEgTWF4LiBMYXVuY2ggY29tcGFyaXNvbjsgcGVyLW1vZGVsIGFnZW50IGhhcm5lc3MsIGV2YWx1YXRpb24gYnVkZ2V0IGFuZCByZXJ1biBwcm92ZW5hbmNlIGFyZSBub3Qgc3BlY2lmaWVkLg","scoreUnit":"percent","url":"https://x.ai/news/grok-4-7","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"healthbench-professional","name":"HealthBench Professional","preferredHarnessId":"healthbench-professional:xai-grok47:R3JvayA0LjcgWEhpZ2ggKERlZXBTV0U6IEhpZ2gpOyBHcm9rIDQuNiBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IENsYXVkZSBGYWJsZSA1LjEgTWF4LiBMYXVuY2ggY29tcGFyaXNvbjsgcGVyLW1vZGVsIGFnZW50IGhhcm5lc3MsIGV2YWx1YXRpb24gYnVkZ2V0IGFuZCByZXJ1biBwcm92ZW5hbmNlIGFyZSBub3Qgc3BlY2lmaWVkLg","scoreUnit":"percent","url":"https://x.ai/news/grok-4-7","version":null},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"frontierswe-2-percent","name":"FrontierSWE · 2","preferredHarnessId":"frontierswe-2-percent:anthropic-opus-5-5-card:UHJveGltYWwgYWdlbnQgaGFybmVzczsgbWF4IHJlYXNvbmluZyBlZmZvcnQ7IDM0IHRhc2tzOyBmaXZlIHRyaWFscyBwZXIgdGFzazsgbWVhbiBhY3Jvc3MgdHJpYWxzLg","scoreUnit":"percent","url":"https://www-cdn.anthropic.com/fc1b44717c85dc068bc6ba5024219938094694bd/Claude%20Opus%205.5%20System%20Card.pdf","version":"2"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"cursorbench-4-0-percent","name":"CursorBench · 4.0","preferredHarnessId":"cursorbench-4-0-percent:anthropic-opus-5-5-card:Q3Vyc29yIHByb2R1Y3Rpb24gYWdlbnQgaGFybmVzczsgbWF4IGVmZm9ydC4gSW5kZXBlbmRlbnRseSBtZWFzdXJlZCBieSBDdXJzb3I7IEFudGhyb3BpYyBlc3RpbWF0ZWQgY29zdCBmcm9tIEN1cnNvciB0b2tlbiBjb3VudHMu","scoreUnit":"percent","url":"https://www-cdn.anthropic.com/fc1b44717c85dc068bc6ba5024219938094694bd/Claude%20Opus%205.5%20System%20Card.pdf","version":"4.0"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"osworld-2-0-september-10-2026-task-release","name":"OSWorld · 2.0 September 10, 2026 task release","preferredHarnessId":"osworld-2-0-september-10-2026-task-release:anthropic-opus-5-5-card:UGFydGlhbCBzY29yZSwgcGFzc0AxOyAxMDggdGFza3M7IGZpdmUgcnVuczsgMTA4MHA7IDUwMCBhY3Rpb24gc3RlcHM7IG1heCByZWFzb25pbmcgZWZmb3J0OyBPcHVzIDQuOCBncmFkZXIgd2hlcmUgcmVxdWlyZWQuIFNlcHRlbWJlciAxMCwgMjAyNiB0YXNrIGZpbGVzIGFuZCBzZXJ2ZXItc2lkZSBjb250ZXh0IGNvbXBhY3Rpb24gYWZ0ZXIgMTAwayB0b2tlbnMu","scoreUnit":"percent","url":"https://www-cdn.anthropic.com/fc1b44717c85dc068bc6ba5024219938094694bd/Claude%20Opus%205.5%20System%20Card.pdf","version":"2.0 September 10, 2026 task release"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"gdpval-aa-2-1","name":"GDPval-AA · 2.1","preferredHarnessId":"gdpval-aa-2-1:anthropic-opus-5-5-card:QXJ0aWZpY2lhbCBBbmFseXNpcyBHRFB2YWwtQUEgdjIuMTsgMjIwIEdEUHZhbCBnb2xkIHRhc2tzOyBibGluZCBwYWlyd2lzZSBFbG8gYW5jaG9yZWQgdG8gRGVlcFNlZWsgVjQuMSBGbGFzaCAobWF4KSBhdCAxNjAwOyBtYXggZWZmb3J0LiBSdW4gYnkgQXJ0aWZpY2lhbCBBbmFseXNpcy4","scoreUnit":"elo","url":"https://www-cdn.anthropic.com/fc1b44717c85dc068bc6ba5024219938094694bd/Claude%20Opus%205.5%20System%20Card.pdf","version":"2.1"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"aa-briefcase-1-1-elo","name":"AA-Briefcase · 1.1","preferredHarnessId":"aa-briefcase-1-1-elo:anthropic-opus-5-5-card:QXJ0aWZpY2lhbCBBbmFseXNpcyBBQS1CcmllZmNhc2UgdjEuMTsgbG9uZy1ob3Jpem9uIGtub3dsZWRnZSBwcm9qZWN0czsgcnVicmljIHNjb3JpbmcgYW5kIHBhaXJ3aXNlIGp1ZGdpbmc7IG1heCBlZmZvcnQuIFJ1biBieSBBcnRpZmljaWFsIEFuYWx5c2lzLg","scoreUnit":"elo","url":"https://www-cdn.anthropic.com/fc1b44717c85dc068bc6ba5024219938094694bd/Claude%20Opus%205.5%20System%20Card.pdf","version":"1.1"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"automationbench-1-0-6-percent","name":"AutomationBench · 1.0.6","preferredHarnessId":"automationbench-1-0-6-percent:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBsb3cuIEF1dG9tYXRpb25CZW5jaCAxLjAuNi4gRW5kLXRvLWVuZCB3b3JrZmxvd3MgYWNyb3NzIDQ3IHRvb2xzIGluIHNhbGVzLCBtYXJrZXRpbmcsIG9wZXJhdGlvbnMsIHN1cHBvcnQsIGZpbmFuY2UsIGFuZCBIUi4gVGhlIHBhZ2Ugc2F5cyB0aGUgQ2xhdWRlIEZhYmxlIDUuMSBjb3N0IHBvaW50IG9taXRzIE9wdXMgNSBmYWxsYmFja3MuIE9wZW5BSS1wdWJsaXNoZWQgY2hhcnQgb24gdGhlIEludHJvZHVjaW5nIEdQVC02IFNvbCBhbmQgTHVuYSBwYWdlLg","scoreUnit":"percent","url":"https://openai.com/index/introducing-gpt-6-sol-and-luna/","version":"1.0.6"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"agents-last-exam-1","name":"Agents' Last Exam · V1","preferredHarnessId":"agents-last-exam-1:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBsb3cuIEFnZW50cycgTGFzdCBFeGFtIFYxLiBMb25nLWhvcml6b24gcHJvZmVzc2lvbmFsIHRhc2tzIGFjcm9zcyA1NSBzdWItaW5kdXN0cmllcy4gT3BlbkFJLXB1Ymxpc2hlZCBjaGFydCBvbiB0aGUgSW50cm9kdWNpbmcgR1BULTYgU29sIGFuZCBMdW5hIHBhZ2Uu","scoreUnit":"percent","url":"https://openai.com/index/introducing-gpt-6-sol-and-luna/","version":"V1"},{"combiner":{"inDefault":false},"higherIsBetter":true,"id":"frontiercode-1-1-main-score-1-1-main","name":"FrontierCode 1.1 Main (score) · 1.1 Main","preferredHarnessId":"frontiercode-1-1-main-score-1-1-main:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBsb3cuIEZyb250aWVyQ29kZSAxLjEgTWFpbiwgYXMgaWRlbnRpZmllZCBpbiB0aGUgc3Vycm91bmRpbmcgYXJ0aWNsZS4gVGhlIGNoYXJ0IHRpdGxlIGlzIEZyb250aWVyQ29kZS4gR3JhZGVkIG9uIGNvcnJlY3RuZXNzIGFuZCBtZXJnZWFiaWxpdHkuIE9wZW5BSS1wdWJsaXNoZWQgY2hhcnQgb24gdGhlIEludHJvZHVjaW5nIEdQVC02IFNvbCBhbmQgTHVuYSBwYWdlLg","scoreUnit":"percent","url":"https://openai.com/index/introducing-gpt-6-sol-and-luna/","version":"1.1 Main"},{"combiner":{"inDefault":false},"higherIsBetter":false,"id":"factual-error-rate-on-difficult-prompts-sol-6-1-launch-user-flagged-conversations","name":"Factual error rate on difficult prompts · Sol 6.1 launch / user-flagged conversations","preferredHarnessId":"factual-error-rate-on-difficult-prompts-sol-6-1-launch-user-flagged-conversations:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBsb3cuIFNoYXJlIG9mIGFuc3dlcnMgd2l0aCBhdCBsZWFzdCBvbmUgZmFjdHVhbCBlcnJvciBvbiBkZS1pZGVudGlmaWVkIGNvbnZlcnNhdGlvbnMgd2hlcmUgdXNlcnMgZmxhZ2dlZCBhbiBlYXJsaWVyIG1vZGVsIGVycm9yOyBkZWxpYmVyYXRlbHkgZGlmZmljdWx0LCBub3QgcmVwcmVzZW50YXRpdmUgb2YgdHlwaWNhbCB1c2FnZS4gT3BlbkFJIHJlc2VhcmNoIGVudmlyb25tZW50IG9yIEFQSTsgc2FtZSBuYW1lZCBtZXRyaWMgYW5kIGV2YWx1YXRpb24gY2hhcnQuIE5vIGZhbGxiYWNrIGlzIHJlcG9ydGVkIGZvciB0aGlzIE9wZW5BSSBjb25maWd1cmF0aW9uLg","scoreUnit":"percent","url":"https://openai.com/index/introducing-gpt-6-1-sol/","version":"Sol 6.1 launch / user-flagged conversations"},{"combiner":{"inDefault":false},"higherIsBetter":false,"id":"failure-to-disclose-a-broken-search-tool-sol-6-1-launch-safety-stress-test","name":"Failure to disclose a broken search tool · Sol 6.1 launch / safety stress test","preferredHarnessId":"failure-to-disclose-a-broken-search-tool-sol-6-1-launch-safety-stress-test:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIEFkdmVyc2FyaWFsIHRlc3Qgb2Ygd2hldGhlciBhZ2VudHMgZGlzY2xvc2UgYSBicm9rZW4gc2VhcmNoIHRvb2wuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","scoreUnit":"percent","url":"https://openai.com/index/introducing-gpt-6-1-sol/","version":"Sol 6.1 launch / safety stress test"},{"combiner":{"inDefault":false},"higherIsBetter":false,"id":"reviewer-bypass-attempts-sol-6-1-launch-safety-stress-test","name":"Reviewer bypass attempts · Sol 6.1 launch / safety stress test","preferredHarnessId":"reviewer-bypass-attempts-sol-6-1-launch-safety-stress-test:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIEF0dGVtcHRzIHRvIHdvcmsgYXJvdW5kIGFuIGF1dG9tYXRlZCBzYWZldHkgcmV2aWV3ZXIgYmxvY2tpbmcgYW4gYWN0aW9uLiBPcGVuQUkgcmVzZWFyY2ggZW52aXJvbm1lbnQgb3IgQVBJOyBzYW1lIG5hbWVkIG1ldHJpYyBhbmQgZXZhbHVhdGlvbiBjaGFydC4gTm8gZmFsbGJhY2sgaXMgcmVwb3J0ZWQgZm9yIHRoaXMgT3BlbkFJIGNvbmZpZ3VyYXRpb24u","scoreUnit":"percent","url":"https://openai.com/index/introducing-gpt-6-1-sol/","version":"Sol 6.1 launch / safety stress test"},{"combiner":{"inDefault":false},"higherIsBetter":false,"id":"warning-circumvention-sol-6-1-launch-safety-stress-test","name":"Warning circumvention · Sol 6.1 launch / safety stress test","preferredHarnessId":"warning-circumvention-sol-6-1-launch-safety-stress-test:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIFJlc3RyaWN0aW9uLWNpcmN1bXZlbnRpb24gYXR0ZW1wdHMgaW4gZGVsaWJlcmF0ZWx5IGRpZmZpY3VsdCBsb3ctc3Rha2VzIGNhc2VzIHdpdGhvdXQgZnVsbCBwcm9kdWN0IHNhZmVndWFyZHMuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","scoreUnit":"percent","url":"https://openai.com/index/introducing-gpt-6-1-sol/","version":"Sol 6.1 launch / safety stress test"},{"combiner":{"inDefault":false},"higherIsBetter":false,"id":"computer-use-safety-stress-test-sol-6-1-launch-safety-stress-test","name":"Computer-use safety stress test · Sol 6.1 launch / safety stress test","preferredHarnessId":"computer-use-safety-stress-test-sol-6-1-launch-safety-stress-test:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCB4aGlnaC4gVW5pbnRlbmRlZCBvdXRjb21lcyBpbiBkZWxpYmVyYXRlbHkgYWR2ZXJzYXJpYWwgY29tcHV0ZXItIGFuZCBicm93c2VyLXVzZSB3b3JrcGxhY2UgdGFza3M7IHRoZSB1cGRhdGVkIGhhcmRlciBzYWZldHkgc3Vic2V0LCBub3QgT1NXb3JsZCB0YXNrIHN1Y2Nlc3MuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","scoreUnit":"percent","url":"https://openai.com/index/introducing-gpt-6-1-sol/","version":"Sol 6.1 launch / safety stress test"}],"harnesses":[{"benchmarkId":"vulcanbench-v3","id":"vulcanbench-v3-07-grok-fable-sol-medium","name":"VulcanBench v3 bare-bones API · Report 07 · medium","notes":"Display only. 23 tasks; report-specific repeats and time budgets. See source report for configuration."},{"benchmarkId":"vulcanbench-v3","id":"vulcanbench-v3-07-grok-fable-sol-high","name":"VulcanBench v3 bare-bones API · Report 07 · high","notes":"Display only. 23 tasks; report-specific repeats and time budgets. See source report for configuration."},{"benchmarkId":"vulcanbench-v3","id":"vulcanbench-v3-07-grok-fable-sol-low","name":"VulcanBench v3 bare-bones API · Report 07 · low","notes":"Display only. 23 tasks; report-specific repeats and time budgets. See source report for configuration."},{"benchmarkId":"vulcanbench-v3","id":"vulcanbench-v3-08-kimi-k3-max","name":"VulcanBench v3 bare-bones API · Report 08 · max","notes":"Display only. 23 tasks; report-specific repeats and time budgets. See source report for configuration."},{"benchmarkId":"vulcanbench-v3","id":"vulcanbench-v3-10-opus5-effort-low","name":"VulcanBench v3 bare-bones API · Report 10 · low","notes":"Display only. 23 tasks; report-specific repeats and time budgets. See source report for configuration."},{"benchmarkId":"vulcanbench-v3","id":"vulcanbench-v3-10-opus5-effort-medium","name":"VulcanBench v3 bare-bones API · Report 10 · medium","notes":"Display only. 23 tasks; report-specific repeats and time budgets. See source report for configuration."},{"benchmarkId":"vulcanbench-v3","id":"vulcanbench-v3-10-opus5-effort-high","name":"VulcanBench v3 bare-bones API · Report 10 · high","notes":"Display only. 23 tasks; report-specific repeats and time budgets. See source report for configuration."},{"benchmarkId":"vulcanbench-v3","id":"vulcanbench-v3-12-qwen38-max-low","name":"VulcanBench v3 bare-bones API · Report 12 · low","notes":"Display only. 23 tasks; report-specific repeats and time budgets. See source report for configuration."},{"benchmarkId":"vulcanbench-v3","id":"vulcanbench-v3-12-qwen38-max-medium","name":"VulcanBench v3 bare-bones API · Report 12 · medium","notes":"Display only. 23 tasks; report-specific repeats and time budgets. See source report for configuration."},{"benchmarkId":"vulcanbench-v3","id":"vulcanbench-v3-12-qwen38-max-xhigh","name":"VulcanBench v3 bare-bones API · Report 12 · xhigh","notes":"Display only. 23 tasks; report-specific repeats and time budgets. Source reports 55.1% and 38/64 completed runs; incomplete trials. Published percentage retained."},{"benchmarkId":"vulcanbench-v3","id":"vulcanbench-v3-14-grok-46-effort-low","name":"VulcanBench v3 bare-bones API · Report 14 · low","notes":"Display only. 23 tasks; report-specific repeats and time budgets. See source report for configuration."},{"benchmarkId":"vulcanbench-v3","id":"vulcanbench-v3-14-grok-46-effort-medium","name":"VulcanBench v3 bare-bones API · Report 14 · medium","notes":"Display only. 23 tasks; report-specific repeats and time budgets. See source report for configuration."},{"benchmarkId":"vulcanbench-v3","id":"vulcanbench-v3-14-grok-46-effort-high","name":"VulcanBench v3 bare-bones API · Report 14 · high","notes":"Display only. 23 tasks; report-specific repeats and time budgets. See source report for configuration."},{"benchmarkId":"vulcanbench-v3","id":"vulcanbench-v3-14-grok-46-effort-xhigh","name":"VulcanBench v3 bare-bones API · Report 14 · xhigh","notes":"Display only. 23 tasks; report-specific repeats and time budgets. See source report for configuration."},{"benchmarkId":"vulcanbench-v3","id":"vulcanbench-v3-17-qwen38-27b-effort-low","name":"VulcanBench v3 bare-bones API · Report 17 · low","notes":"Display only. 23 tasks; report-specific repeats and time budgets. See source report for configuration."},{"benchmarkId":"vulcanbench-v3","id":"vulcanbench-v3-17-qwen38-27b-effort-medium","name":"VulcanBench v3 bare-bones API · Report 17 · medium","notes":"Display only. 23 tasks; report-specific repeats and time budgets. See source report for configuration."},{"benchmarkId":"vulcanbench-v3","id":"vulcanbench-v3-17-qwen38-27b-effort-xhigh","name":"VulcanBench v3 bare-bones API · Report 17 · xhigh","notes":"Display only. 23 tasks; report-specific repeats and time budgets. See source report for configuration."},{"benchmarkId":"vulcanbench-v3","id":"vulcanbench-v3-18-glm53-zcode-harness-low","name":"VulcanBench v3 bare-bones API · Report 18 · low","notes":"Display only. 23 tasks; report-specific repeats and time budgets. Source labels low/high/max; Z.ai API effort labels may be metadata only. Bare-bones API column, not ZCode."},{"benchmarkId":"vulcanbench-v3","id":"vulcanbench-v3-18-glm53-zcode-harness-high","name":"VulcanBench v3 bare-bones API · Report 18 · high","notes":"Display only. 23 tasks; report-specific repeats and time budgets. Source labels low/high/max; Z.ai API effort labels may be metadata only. Bare-bones API column, not ZCode."},{"benchmarkId":"vulcanbench-v3","id":"vulcanbench-v3-18-glm53-zcode-harness-max","name":"VulcanBench v3 bare-bones API · Report 18 · max","notes":"Display only. 23 tasks; report-specific repeats and time budgets. Source labels low/high/max; Z.ai API effort labels may be metadata only. Bare-bones API column, not ZCode."},{"benchmarkId":"vulcanbench-v3","id":"vulcanbench-v3-19-musespark-effort-low","name":"VulcanBench v3 bare-bones API · Report 19 · low","notes":"Display only. 23 tasks; report-specific repeats and time budgets. See source report for configuration."},{"benchmarkId":"vulcanbench-v3","id":"vulcanbench-v3-19-musespark-effort-high","name":"VulcanBench v3 bare-bones API · Report 19 · high","notes":"Display only. 23 tasks; report-specific repeats and time budgets. See source report for configuration."},{"benchmarkId":"vulcanbench-v3","id":"vulcanbench-v3-19-musespark-effort-xhigh","name":"VulcanBench v3 bare-bones API · Report 19 · xhigh","notes":"Display only. 23 tasks; report-specific repeats and time budgets. See source report for configuration."},{"benchmarkId":"autoresearchexam","id":"autoresearchexam-terminus-2","name":"AutoResearchExam Terminus 2","notes":"Display only. Hidden-test AUARC, 29 tasks, 24-hour Terminus 2 runs. Excluded from weighted Overall."},{"benchmarkId":"bughunt-bench","id":"bughunt-bench-max","name":"Bug Hunt Bench · max","notes":"Display only. Percent of 105 planted bugs fixed. Featured, non-superseded rows. Not in the weighted ranking."},{"benchmarkId":"bughunt-bench","id":"bughunt-bench-xhigh","name":"Bug Hunt Bench · xhigh","notes":"Display only. Percent of 105 planted bugs fixed. Featured, non-superseded rows. Not in the weighted ranking."},{"benchmarkId":"bughunt-bench","id":"bughunt-bench-high","name":"Bug Hunt Bench · high","notes":"Display only. Percent of 105 planted bugs fixed. Featured, non-superseded rows. Not in the weighted ranking."},{"benchmarkId":"bughunt-bench","id":"bughunt-bench-default","name":"Bug Hunt Bench · default","notes":"Display only. Percent of 105 planted bugs fixed. Featured, non-superseded rows. Not in the weighted ranking."},{"benchmarkId":"real-swe","id":"real-swe-fable-5-1-claude-code","name":"Real-SWE · Fable 5.1 · Claude Code","notes":"Display only. One model plus its native harness, high reasoning, 10 private tasks, 8 runs. Not a bare-model score. Excluded from weighted Overall."},{"benchmarkId":"real-swe","id":"real-swe-gpt-6-astra-codex-cli","name":"Real-SWE · GPT-6 Astra · Codex CLI","notes":"Display only. One model plus its native harness, high reasoning, 10 private tasks, 8 runs. Not a bare-model score. Excluded from weighted Overall."},{"benchmarkId":"real-swe","id":"real-swe-grok-4-6-grok-build","name":"Real-SWE · Grok 4.6 · Grok Build","notes":"Display only. One model plus its native harness, high reasoning, 10 private tasks, 8 runs. Not a bare-model score. Excluded from weighted Overall."},{"benchmarkId":"real-swe","id":"real-swe-gemini-3-8-flash-gemini-cli","name":"Real-SWE · Gemini 3.8 Flash · Gemini CLI","notes":"Display only. One model plus its native harness, high reasoning, 10 private tasks, 8 runs. Not a bare-model score. Excluded from weighted Overall."},{"benchmarkId":"real-swe","id":"real-swe-glm-5-3-claude-code","name":"Real-SWE · GLM 5.3 · Claude Code","notes":"Display only. One model plus its native harness, high reasoning, 10 private tasks, 8 runs. Not a bare-model score. Excluded from weighted Overall."},{"benchmarkId":"real-swe","id":"real-swe-muse-spark-1-3-muse-code","name":"Real-SWE · Muse Spark 1.3 · Muse Code","notes":"Display only. One model plus its native harness, high reasoning, 10 private tasks, 8 runs. Not a bare-model score. Excluded from weighted Overall."},{"benchmarkId":"real-swe","id":"real-swe-kimi-k3-kimi-code","name":"Real-SWE · Kimi K3 · Kimi Code","notes":"Display only. One model plus its native harness, high reasoning, 10 private tasks, 8 runs. Not a bare-model score. Excluded from weighted Overall."},{"benchmarkId":"real-swe","id":"real-swe-gpt-5-6-sol-codex-cli","name":"Real-SWE · GPT-5.6 Sol · Codex CLI","notes":"Display only. One model plus its native harness, high reasoning, 10 private tasks, 8 runs. Not a bare-model score. Excluded from weighted Overall."},{"benchmarkId":"swe-bench-verified","id":"swe-bench-verified-official","name":"SWE-bench Verified official","notes":"Preferred ranking harness. Official board rows are mini-SWE-agent. Lab-card rows stay until the dump names that model."},{"benchmarkId":"swe-bench-pro","id":"swe-bench-pro-official","name":"SWE-bench Pro reported","notes":"Preferred ranking harness. Official public-board rows are mini-swe-agent. Lab-card rows stay until the dump names that model. Private Scale set is out of scope."},{"benchmarkId":"terminal-bench-2.1","id":"terminal-bench-2.1-reported","name":"Terminal-Bench 2.1 reported","notes":"Preferred ranking harness. Official board rows are Terminus 2. Lab-card rows stay until the dump names that model."},{"benchmarkId":"terminal-bench-2.1","id":"terminal-bench-2.1-codex-ultra","name":"Codex ultra","notes":"OpenAI Sol ultra subagent mode. Not the default ranking harness."},{"benchmarkId":"terminal-bench-4.0","id":"terminal-bench-4.0-reported","name":"Terminal-Bench 4.0 reported","notes":"Weighted coding harness. Official Harbor 4-0-0 display rows. Native agents are not a shared Terminus 2 harness. One best-accuracy row per catalog model."},{"benchmarkId":"deepswe-v1.1","id":"deepswe-v1.1-reported","name":"DeepSWE v1.1 reported","notes":null},{"benchmarkId":"osworld-verified","id":"osworld-verified-reported","name":"OSWorld-Verified reported","notes":null},{"benchmarkId":"gdpval-aa","id":"gdpval-aa-official","name":"Artificial Analysis GDPval-AA","notes":"Official AA Elo."},{"benchmarkId":"automationbench-aa","id":"automationbench-aa-official","name":"Artificial Analysis AutomationBench-AA","notes":"AA score: share of task objectives completed with no guardrail violation. Private 657-task holdout. Not Zapier tasks completed."},{"benchmarkId":"gpqa-diamond","id":"gpqa-diamond-reported","name":"GPQA Diamond reported","notes":"No live official board. Lab cards and independent runs."},{"benchmarkId":"hle","id":"hle-no-tools","name":"HLE no tools","notes":"Preferred default harness."},{"benchmarkId":"hle","id":"hle-with-tools","name":"HLE with tools","notes":"Not the default ranking harness."},{"benchmarkId":"arc-agi-2","id":"arc-agi-2-official","name":"ARC Prize verified","notes":null},{"benchmarkId":"livecodebench","id":"livecodebench-reported","name":"LiveCodeBench pass@1","notes":"Not LiveCodeBench Pro Elo."},{"benchmarkId":"lmarena-text","id":"lmarena-text-official","name":"LMArena Text","notes":null},{"benchmarkId":"mmlu-pro","id":"mmlu-pro-reported","name":"MMLU-Pro reported","notes":null},{"benchmarkId":"aa-intelligence-index","id":"aa-intelligence-index-official","name":"AA Intelligence Index","notes":"Composite. Preset only."},{"benchmarkId":"browsecomp","id":"browsecomp-reported","name":"BrowseComp reported","notes":null},{"benchmarkId":"tau2-telecom","id":"tau2-telecom-reported","name":"τ²-bench Telecom reported","notes":null},{"benchmarkId":"paperbench","id":"paperbench-reported","name":"PaperBench reported","notes":null},{"benchmarkId":"swe-bench-pro","id":"swe-bench-pro-anthropic-reported","name":"Anthropic SWE-bench Pro reported system","notes":"First-party custom or unspecified agent configuration; not assumed equivalent to preferred agent. Source: https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"deepswe-v1.1","id":"deepswe-v1.1-anthropic-reported","name":"Anthropic DeepSWE1.1 reported system","notes":"First-party custom or unspecified agent configuration; not assumed equivalent to preferred agent. Source: https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"swe-bench-pro-percent-higher","id":"swe-bench-pro-percent-higher:anthropic-fable-5-1-card:QW50aHJvcGljIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IGFkYXB0aXZlIHRoaW5raW5nIGF0IG1heCBlZmZvcnQsIGRlZmF1bHQgc2FtcGxpbmcsIGZpdmUtdHJpYWwgbWVhbiB1bmxlc3Mgc2VjdGlvbiBzcGVjaWZpZXMgb3RoZXJ3aXNlOyBjb250ZXh0IGF0IG1vc3QgMU0uIENvbXBhcmF0b3IgY29uZmlndXJhdGlvbiBmb2xsb3dzIGNpdGVkIHByaW9yIGNhcmQgb3IgYm9hcmQuIEFnZW50IGlkZW50aXR5IGlzIG5vdCBlc3RhYmxpc2hlZCBhcyBtaW5pLXN3ZS1hZ2VudC4","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"Anthropic reported configuration; adaptive thinking at max effort, default sampling, five-trial mean unless section specifies otherwise; context at most 1M. Comparator configuration follows cited prior card or board. Agent identity is not established as mini-swe-agent."},{"benchmarkId":"swe-bench-multilingual-percent","id":"swe-bench-multilingual-percent:anthropic-fable-5-1-card:QW50aHJvcGljIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IGFkYXB0aXZlIHRoaW5raW5nIGF0IG1heCBlZmZvcnQsIGRlZmF1bHQgc2FtcGxpbmcsIGZpdmUtdHJpYWwgbWVhbiB1bmxlc3Mgc2VjdGlvbiBzcGVjaWZpZXMgb3RoZXJ3aXNlOyBjb250ZXh0IGF0IG1vc3QgMU0uIENvbXBhcmF0b3IgY29uZmlndXJhdGlvbiBmb2xsb3dzIGNpdGVkIHByaW9yIGNhcmQgb3IgYm9hcmQuIEFnZW50IGlkZW50aXR5IGlzIG5vdCBlc3RhYmxpc2hlZCBhcyBtaW5pLXN3ZS1hZ2VudC4","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"Anthropic reported configuration; adaptive thinking at max effort, default sampling, five-trial mean unless section specifies otherwise; context at most 1M. Comparator configuration follows cited prior card or board. Agent identity is not established as mini-swe-agent."},{"benchmarkId":"swe-bench-multimodal","id":"swe-bench-multimodal:anthropic-fable-5-1-card:QW50aHJvcGljIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IGFkYXB0aXZlIHRoaW5raW5nIGF0IG1heCBlZmZvcnQsIGRlZmF1bHQgc2FtcGxpbmcsIGZpdmUtdHJpYWwgbWVhbiB1bmxlc3Mgc2VjdGlvbiBzcGVjaWZpZXMgb3RoZXJ3aXNlOyBjb250ZXh0IGF0IG1vc3QgMU0uIENvbXBhcmF0b3IgY29uZmlndXJhdGlvbiBmb2xsb3dzIGNpdGVkIHByaW9yIGNhcmQgb3IgYm9hcmQuIEFnZW50IGlkZW50aXR5IGlzIG5vdCBlc3RhYmxpc2hlZCBhcyBtaW5pLXN3ZS1hZ2VudC4","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"Anthropic reported configuration; adaptive thinking at max effort, default sampling, five-trial mean unless section specifies otherwise; context at most 1M. Comparator configuration follows cited prior card or board. Agent identity is not established as mini-swe-agent."},{"benchmarkId":"deepswe-1-1-percent","id":"deepswe-1-1-percent:anthropic-fable-5-1-card:QW50aHJvcGljIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IGFkYXB0aXZlIHRoaW5raW5nIGF0IG1heCBlZmZvcnQsIGRlZmF1bHQgc2FtcGxpbmcsIGZpdmUtdHJpYWwgbWVhbiB1bmxlc3Mgc2VjdGlvbiBzcGVjaWZpZXMgb3RoZXJ3aXNlOyBjb250ZXh0IGF0IG1vc3QgMU0uIENvbXBhcmF0b3IgY29uZmlndXJhdGlvbiBmb2xsb3dzIGNpdGVkIHByaW9yIGNhcmQgb3IgYm9hcmQuIDExMyB0YXNrczsgb3JpZ2luYWwgaGlkZGVuLXRlc3QgZ3JhZGluZy4","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"Anthropic reported configuration; adaptive thinking at max effort, default sampling, five-trial mean unless section specifies otherwise; context at most 1M. Comparator configuration follows cited prior card or board. 113 tasks; original hidden-test grading."},{"benchmarkId":"frontiercode-1-1-extended","id":"frontiercode-1-1-extended:anthropic-fable-5-1-card:Q29nbml0aW9uIGFnZW50aWMgY29kaW5nOyBjb21wb3NpdGUgZnVuY3Rpb25hbCBhbmQgY29kZS1xdWFsaXR5IHNjb3JlLiBGYWJsZTUuMSBtZWRpdW0gZWZmb3J0OyBGYWJsZTUgeGhpZ2gu","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"Cognition agentic coding; composite functional and code-quality score. Fable5.1 medium effort; Fable5 xhigh."},{"benchmarkId":"frontiercode-1-1-main","id":"frontiercode-1-1-main:anthropic-fable-5-1-card:Q29nbml0aW9uIGFnZW50aWMgY29kaW5nOyBjb21wb3NpdGUgZnVuY3Rpb25hbCBhbmQgY29kZS1xdWFsaXR5IHNjb3JlLiBGYWJsZTUuMSBtZWRpdW0gZWZmb3J0OyBGYWJsZTUgeGhpZ2gu","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"Cognition agentic coding; composite functional and code-quality score. Fable5.1 medium effort; Fable5 xhigh."},{"benchmarkId":"frontierswe-2","id":"frontierswe-2:anthropic-fable-5-1-card:UHJveGltYWwgYWdlbnQgaGFybmVzczsgbWF4IGVmZm9ydDsgMzQgdGFza3MsIGZpdmUgdHJpYWxzL3Rhc2s7IG1lYW4gc2NvcmUgb24gMC4uMSBzY2FsZS4","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"Proximal agent harness; max effort; 34 tasks, five trials/task; mean score on 0..1 scale."},{"benchmarkId":"terminal-bench-4-0-percent","id":"terminal-bench-4-0-percent:anthropic-fable-5-1-card:Q2xhdWRlIENvZGUgLS1iYXJlIG1heCBlZmZvcnQsIDE1IHRyaWFscy90YXNrIG92ZXIgNjYgdGFza3M7IEFudGhyb3BpYyBpbnRlcm5hbCByZXJ1bnMu","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"Claude Code --bare max effort, 15 trials/task over 66 tasks; Anthropic internal reruns."},{"benchmarkId":"terminal-bench-4-0-percent","id":"terminal-bench-4-0-percent:anthropic-fable-5-1-card:Q2xhdWRlIENvZGUgLS1iYXJlIG1heCBlZmZvcnQsIDE1IHRyaWFscy90YXNrIG92ZXI2NnRhc2tzIGZvciBDbGF1ZGU7IEdQVCBDb2RleCBDTEkgbWF4IGZyb20gcHVibGljIGJvYXJkLg","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"Claude Code --bare max effort, 15 trials/task over66tasks for Claude; GPT Codex CLI max from public board."},{"benchmarkId":"terminal-bench-science-0-1-percent","id":"terminal-bench-science-0-1-percent:anthropic-fable-5-1-card:NzB0YXNrczsgQ2xhdWRlIENvZGUgLS1iYXJlIG1heDsgRmFibGUxMHRyaWFscy90YXNrLCBPcHVzMTI7IEdQVCBDb2RleCBDTEkgbWF4IGZyb20gcHVibGljIGJvYXJkLg","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"70tasks; Claude Code --bare max; Fable10trials/task, Opus12; GPT Codex CLI max from public board."},{"benchmarkId":"cursorbench-3-2-0","id":"cursorbench-3-2-0:anthropic-fable-5-1-card:Q3Vyc29yIHByb2R1Y3Rpb24gYWdlbnQgaGFybmVzczsgaW5kZXBlbmRlbnRseSBtZWFzdXJlZCBieSBDdXJzb3I7IG1heCBlZmZvcnQu","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"Cursor production agent harness; independently measured by Cursor; max effort."},{"benchmarkId":"cursorbench-3-2-0","id":"cursorbench-3-2-0:anthropic-fable-5-1-card:Q3Vyc29yIHByb2R1Y3Rpb24gYWdlbnQgaGFybmVzczsgbWVkaXVtIGVmZm9ydDsgJDMuNTMvdGFzay4","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"Cursor production agent harness; medium effort; $3.53/task."},{"benchmarkId":"programbench-166-golden-task-subset","id":"programbench-166-golden-task-subset:anthropic-fable-5-1-card:bWluaS1zd2UtYWdlbnQgd2l0aG91dCBzaXgtaG91ciB0aW1lb3V0OyBleGNsdWRlczM0Zmxha3ktcmVmZXJlbmNlIHRhc2tzOyB0ZXN0cyByZXN0cmljdGVkIHRvIHJlZmVyZW5jZS1wYXNzaW5nIHRlc3RzOyB1cCB0bzFNY29udGV4dC4","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"mini-swe-agent without six-hour timeout; excludes34flaky-reference tasks; tests restricted to reference-passing tests; up to1Mcontext."},{"benchmarkId":"humanity-s-last-exam","id":"humanity-s-last-exam:anthropic-fable-5-1-card:RnVsbDI1MDBxdWVzdGlvbnM7IG5vIHRvb2xzOyBhdXRvIHRoaW5raW5nOzFNdG90YWwgdG9rZW4gY2FwOyBubyBjb21wYWN0aW9uOyBPcHVzNC42Z3JhZGVyOyByZXN0cmljdGVkIGZldGNoIGFuZCBjb250YW1pbmF0aW9uIHJldmlldyBmb3IgdG9vbHMu","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"Full2500questions; no tools; auto thinking;1Mtotal token cap; no compaction; Opus4.6grader; restricted fetch and contamination review for tools."},{"benchmarkId":"humanity-s-last-exam","id":"humanity-s-last-exam:anthropic-fable-5-1-card:RnVsbDI1MDBxdWVzdGlvbnM7IHdpdGggdG9vbHM7IGF1dG8gdGhpbmtpbmc7MU10b3RhbCB0b2tlbiBjYXA7IG5vIGNvbXBhY3Rpb247IE9wdXM0LjZncmFkZXI7IHJlc3RyaWN0ZWQgZmV0Y2ggYW5kIGNvbnRhbWluYXRpb24gcmV2aWV3IGZvciB0b29scy4","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"Full2500questions; with tools; auto thinking;1Mtotal token cap; no compaction; Opus4.6grader; restricted fetch and contamination review for tools."},{"benchmarkId":"chartography","id":"chartography:anthropic-fable-5-1-card:MTAwdGFza3M7IGFkYXB0aXZlIHRoaW5raW5nIG1heDsgZml2ZSBydW5zOyBubyB0b29sczsgdG9vbHMgY29uZGl0aW9uIGhhcyBjb250YWluZXIsIHN0YW5kYXJkIGxpYnJhcmllcyBhbmQgY3JvcCB0b29sLg","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"100tasks; adaptive thinking max; five runs; no tools; tools condition has container, standard libraries and crop tool."},{"benchmarkId":"chartography","id":"chartography:anthropic-fable-5-1-card:MTAwdGFza3M7IGFkYXB0aXZlIHRoaW5raW5nIG1heDsgZml2ZSBydW5zOyB3aXRoIHRvb2xzOyB0b29scyBjb25kaXRpb24gaGFzIGNvbnRhaW5lciwgc3RhbmRhcmQgbGlicmFyaWVzIGFuZCBjcm9wIHRvb2wu","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"100tasks; adaptive thinking max; five runs; with tools; tools condition has container, standard libraries and crop tool."},{"benchmarkId":"benchcad-vision2code-1000-file-subset","id":"benchcad-vision2code-1000-file-subset:anthropic-fable-5-1-card:UmFuZG9tMTAwMCBvZjE3OTAwZmlsZXM7IGZpdmUgcnVuczsgYWRhcHRpdmUgdGhpbmtpbmcgbWF4OyBubyB0b29sczsgY29ycmVjdGVkIGNhbWVyYSBwcm9tcHQsIHJhdyBzaGFwZXMgYWNjZXB0ZWQsIGxhc3QgY29kZSBmZW5jZSBwYXJzZWQu","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"Random1000 of17900files; five runs; adaptive thinking max; no tools; corrected camera prompt, raw shapes accepted, last code fence parsed."},{"benchmarkId":"benchcad-vision2code-1000-file-subset","id":"benchcad-vision2code-1000-file-subset:anthropic-fable-5-1-card:UmFuZG9tMTAwMCBvZjE3OTAwZmlsZXM7IGZpdmUgcnVuczsgYWRhcHRpdmUgdGhpbmtpbmcgbWF4OyB3aXRoIHRvb2xzOyBjb3JyZWN0ZWQgY2FtZXJhIHByb21wdCwgcmF3IHNoYXBlcyBhY2NlcHRlZCwgbGFzdCBjb2RlIGZlbmNlIHBhcnNlZC4","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"Random1000 of17900files; five runs; adaptive thinking max; with tools; corrected camera prompt, raw shapes accepted, last code fence parsed."},{"benchmarkId":"osworld-2-0-august2026-task-release","id":"osworld-2-0-august2026-task-release:anthropic-fable-5-1-card:cGFydGlhbCBwYXNzQDE7MTA4dGFza3M7IGZpdmUgcnVuczsxMDgwcDs1MDBzdGVwczsgbWF4IGVmZm9ydDsgT3B1czQuOGdyYWRlcjsgdGFzayBmaXhlczsgRmFibGUgc2FmZXR5IGludGVydmVudGlvbnMgc2NvcmUgemVyby4","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"partial pass@1;108tasks; five runs;1080p;500steps; max effort; Opus4.8grader; task fixes; Fable safety interventions score zero."},{"benchmarkId":"osworld-2-0-august2026-task-release","id":"osworld-2-0-august2026-task-release:anthropic-fable-5-1-card:c3RyaWN0IHBhc3NAMTsxMDh0YXNrczsgZml2ZSBydW5zOzEwODBwOzUwMHN0ZXBzOyBtYXggZWZmb3J0OyBPcHVzNC44Z3JhZGVyOyB0YXNrIGZpeGVzOyBGYWJsZSBzYWZldHkgaW50ZXJ2ZW50aW9ucyBzY29yZSB6ZXJvLg","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"strict pass@1;108tasks; five runs;1080p;500steps; max effort; Opus4.8grader; task fixes; Fable safety interventions score zero."},{"benchmarkId":"officeqa","id":"officeqa:anthropic-fable-5-1-card:RXh0cmFjdGVkLXRleHQgVHJlYXN1cnkgY29ycHVzIGluIHNhbmRib3g7IGNvZGUgZXhlY3V0aW9uOyBwcm9kdWN0aW9uIE1lc3NhZ2VzIEFQSSB3aXRoIHNhZmVndWFyZHMvZmFsbGJhY2s7MTI4a291dHB1dCBjYXAuIFBybzEzM3F1ZXN0aW9ucy4","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"Extracted-text Treasury corpus in sandbox; code execution; production Messages API with safeguards/fallback;128koutput cap. Pro133questions."},{"benchmarkId":"officeqa-pro","id":"officeqa-pro:anthropic-fable-5-1-card:RXh0cmFjdGVkLXRleHQgVHJlYXN1cnkgY29ycHVzIGluIHNhbmRib3g7IGNvZGUgZXhlY3V0aW9uOyBwcm9kdWN0aW9uIE1lc3NhZ2VzIEFQSSB3aXRoIHNhZmVndWFyZHMvZmFsbGJhY2s7MTI4a291dHB1dCBjYXAuIFBybzEzM3F1ZXN0aW9ucy4","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"Extracted-text Treasury corpus in sandbox; code execution; production Messages API with safeguards/fallback;128koutput cap. Pro133questions."},{"benchmarkId":"officeqa-pro","id":"officeqa-pro:anthropic-fable-5-1-card:RGF0YWJyaWNrcyBldmFsdWF0aW9uIHJlYWRpbmcgZG9jdW1lbnRzIGFzIGltYWdlczsgZGlmZmVycyBmcm9tIGV4dHJhY3RlZC10ZXh0IGhhcm5lc3Mu","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"Databricks evaluation reading documents as images; differs from extracted-text harness."},{"benchmarkId":"legal-agent-benchmark-1235-task-public-subset","id":"legal-agent-benchmark-1235-task-public-subset:anthropic-fable-5-1-card:YWxsLXBhc3M7IGZpdmUgcnVuczsgYWRhcHRpdmUgbWF4OyBpbnRlcm5hbCBiYXNoL1B5dGhvbiBoYXJuZXNzLCBTb25uZXQ0LjZqdWRnZTsxNmRlZmVjdGl2ZSB0YXNrcyBleGNsdWRlZDsgcHJvZHVjdGlvbiBzYWZlZ3VhcmRzL2ZhbGxiYWNrLg","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"all-pass; five runs; adaptive max; internal bash/Python harness, Sonnet4.6judge;16defective tasks excluded; production safeguards/fallback."},{"benchmarkId":"legal-agent-benchmark-1235-task-public-subset","id":"legal-agent-benchmark-1235-task-public-subset:anthropic-fable-5-1-card:Y3JpdGVyaW9uLXBhc3M7IGZpdmUgcnVuczsgYWRhcHRpdmUgbWF4OyBpbnRlcm5hbCBiYXNoL1B5dGhvbiBoYXJuZXNzLCBTb25uZXQ0LjZqdWRnZTsxNmRlZmVjdGl2ZSB0YXNrcyBleGNsdWRlZDsgcHJvZHVjdGlvbiBzYWZlZ3VhcmRzL2ZhbGxiYWNrLg","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"criterion-pass; five runs; adaptive max; internal bash/Python harness, Sonnet4.6judge;16defective tasks excluded; production safeguards/fallback."},{"benchmarkId":"legal-agent-benchmark-120-task-held-out-subset","id":"legal-agent-benchmark-120-task-held-out-subset:anthropic-fable-5-1-card:YWxsLXBhc3M7IEFydGlmaWNpYWwgQW5hbHlzaXMgaGFybmVzczsgeGhpZ2ggZWZmb3J0Lg","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"all-pass; Artificial Analysis harness; xhigh effort."},{"benchmarkId":"legal-agent-benchmark-120-task-held-out-subset","id":"legal-agent-benchmark-120-task-held-out-subset:anthropic-fable-5-1-card:Y3JpdGVyaW9uLXBhc3M7IEFydGlmaWNpYWwgQW5hbHlzaXMgaGFybmVzczsgeGhpZ2ggZWZmb3J0Lg","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"criterion-pass; Artificial Analysis harness; xhigh effort."},{"benchmarkId":"gdpval-aa-2","id":"gdpval-aa-2:anthropic-fable-5-1-card:QXJ0aWZpY2lhbCBBbmFseXNpcyBpbmRlcGVuZGVudCBhZ2VudGljIHNoZWxsL3dlYiBldmFsdWF0aW9uOzIyMEdEUHZhbGdvbGR0YXNrczsgYmxpbmQgcGFpcndpc2UgRWxvOyBtYXggZWZmb3J0IGZvciBDbGF1ZGUu","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"Artificial Analysis independent agentic shell/web evaluation;220GDPvalgoldtasks; blind pairwise Elo; max effort for Claude."},{"benchmarkId":"gdpval-aa-2","id":"gdpval-aa-2:anthropic-fable-5-1-card:QXJ0aWZpY2lhbCBBbmFseXNpczsgeGhpZ2ggZWZmb3J0OyBzYW1lIHJlbGVhc2UgYm9hcmQgc25hcHNob3Qu","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"Artificial Analysis; xhigh effort; same release board snapshot."},{"benchmarkId":"aa-briefcase","id":"aa-briefcase:anthropic-fable-5-1-card:QXJ0aWZpY2lhbCBBbmFseXNpcyBsb25nLWhvcml6b24ga25vd2xlZGdlIHByb2plY3RzOyBydWJyaWMgYW5kIHBhbmVsIHBhaXJ3aXNlIGp1ZGdpbmc7IENsYXVkZSBtYXggZWZmb3J0Lg","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"Artificial Analysis long-horizon knowledge projects; rubric and panel pairwise judging; Claude max effort."},{"benchmarkId":"aa-briefcase","id":"aa-briefcase:anthropic-fable-5-1-card:QXJ0aWZpY2lhbCBBbmFseXNpczsgeGhpZ2ggZWZmb3J0Lg","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"Artificial Analysis; xhigh effort."},{"benchmarkId":"aa-briefcase","id":"aa-briefcase:anthropic-fable-5-1-card:QXJ0aWZpY2lhbCBBbmFseXNpczsgaGlnaCBlZmZvcnQu","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"Artificial Analysis; high effort."},{"benchmarkId":"aa-briefcase-percent","id":"aa-briefcase-percent:anthropic-fable-5-1-card:QXJ0aWZpY2lhbCBBbmFseXNpczsgbWF4IGVmZm9ydDsgY29tcG9uZW50IHJ1YnJpYyBwYXNzLg","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"Artificial Analysis; max effort; component rubric pass."},{"benchmarkId":"aa-briefcase-rating","id":"aa-briefcase-rating:anthropic-fable-5-1-card:QXJ0aWZpY2lhbCBBbmFseXNpczsgbWF4IGVmZm9ydDsgY29tcG9uZW50IGFuYWx5dGljYWwgcXVhbGl0eS4","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"Artificial Analysis; max effort; component analytical quality."},{"benchmarkId":"aa-briefcase-rating","id":"aa-briefcase-rating:anthropic-fable-5-1-card:QXJ0aWZpY2lhbCBBbmFseXNpczsgbWF4IGVmZm9ydDsgY29tcG9uZW50IHByZXNlbnRhdGlvbi4","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"Artificial Analysis; max effort; component presentation."},{"benchmarkId":"toolathlon-verified-june2026","id":"toolathlon-verified-june2026:anthropic-fable-5-1-card:UGFzc0AxOzEwOHRhc2tzL3RocmVlIHRyaWFsczsgaW50ZXJuYWwgaGFybmVzczsgbWF4IGVmZm9ydDsgRmFibGUgc2FmZWd1YXJkcytPcHVzNC44ZmFsbGJhY2s7IE9wdXMgc2FmZWd1YXJkcy9mYWxsYmFjayBkaXNhYmxlZDsgcGlubmVkIGNvbnRhaW5lcnMvZGF0YS4","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"Pass@1;108tasks/three trials; internal harness; max effort; Fable safeguards+Opus4.8fallback; Opus safeguards/fallback disabled; pinned containers/data."},{"benchmarkId":"toolathlon-verified-june2026","id":"toolathlon-verified-june2026:anthropic-fable-5-1-card:UGFzc0AzOzEwOHRhc2tzL3RocmVlIHRyaWFsczsgaW50ZXJuYWwgaGFybmVzczsgbWF4IGVmZm9ydDsgRmFibGUgc2FmZWd1YXJkcytPcHVzNC44ZmFsbGJhY2s7IE9wdXMgc2FmZWd1YXJkcy9mYWxsYmFjayBkaXNhYmxlZDsgcGlubmVkIGNvbnRhaW5lcnMvZGF0YS4","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"Pass@3;108tasks/three trials; internal harness; max effort; Fable safeguards+Opus4.8fallback; Opus safeguards/fallback disabled; pinned containers/data."},{"benchmarkId":"toolathlon-verified-june2026","id":"toolathlon-verified-june2026:anthropic-fable-5-1-card:UGFzc8KzOzEwOHRhc2tzL3RocmVlIHRyaWFsczsgaW50ZXJuYWwgaGFybmVzczsgbWF4IGVmZm9ydDsgRmFibGUgc2FmZWd1YXJkcytPcHVzNC44ZmFsbGJhY2s7IE9wdXMgc2FmZWd1YXJkcy9mYWxsYmFjayBkaXNhYmxlZDsgcGlubmVkIGNvbnRhaW5lcnMvZGF0YS4","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"Pass³;108tasks/three trials; internal harness; max effort; Fable safeguards+Opus4.8fallback; Opus safeguards/fallback disabled; pinned containers/data."},{"benchmarkId":"toolathlon-verified-june2026-turns","id":"toolathlon-verified-june2026-turns:anthropic-fable-5-1-card:YXZlcmFnZSB0dXJuczsxMDh0YXNrcy90aHJlZSB0cmlhbHM7IGludGVybmFsIGhhcm5lc3M7IG1heCBlZmZvcnQ7IEZhYmxlIHNhZmVndWFyZHMrT3B1czQuOGZhbGxiYWNrOyBPcHVzIHNhZmVndWFyZHMvZmFsbGJhY2sgZGlzYWJsZWQ7IHBpbm5lZCBjb250YWluZXJzL2RhdGEu","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"average turns;108tasks/three trials; internal harness; max effort; Fable safeguards+Opus4.8fallback; Opus safeguards/fallback disabled; pinned containers/data."},{"benchmarkId":"automationbench","id":"automationbench:anthropic-fable-5-1-card:UHJpdmF0ZSBoZWxkLW91dCBib2FyZDsgc2ltdWxhdGVkIGJ1c2luZXNzLXdvcmtmbG93IGFwcCBBUElzOyBhbGwgYXNzZXJ0aW9ucyBtdXN0IHBhc3M7IG1heCBlZmZvcnQgc3RhdGVkIGZvciBGYWJsZTUuMS9PcHVzNS4","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"Private held-out board; simulated business-workflow app APIs; all assertions must pass; max effort stated for Fable5.1/Opus5."},{"benchmarkId":"arc-agi-1-percent","id":"arc-agi-1-percent:anthropic-fable-5-1-card:QVJDIFByaXplIHNlbWktcHJpdmF0ZSB2YWxpZGF0aW9uOyB2ZXJpZmllZCBGYWJsZTUuMSBtYXggZWZmb3J0OyBjb21wYXJhdG9yIGZpZ3VyZXMgYXMgcmVwb3J0ZWQgaW4gc3VtbWFyeS4","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"ARC Prize semi-private validation; verified Fable5.1 max effort; comparator figures as reported in summary."},{"benchmarkId":"arc-agi-2-percent-higher","id":"arc-agi-2-percent-higher:anthropic-fable-5-1-card:QVJDIFByaXplIHNlbWktcHJpdmF0ZSB2YWxpZGF0aW9uOyB2ZXJpZmllZCBGYWJsZTUuMSBtYXggZWZmb3J0OyBjb21wYXJhdG9yIGZpZ3VyZXMgYXMgcmVwb3J0ZWQgaW4gc3VtbWFyeS4","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"ARC Prize semi-private validation; verified Fable5.1 max effort; comparator figures as reported in summary."},{"benchmarkId":"healthbench","id":"healthbench:anthropic-fable-5-1-card:UmF3IHJ1YnJpYyBzY29yZTsgYWRhcHRpdmUgbWF4OyBmaXZlIHRyaWFsczsgbm8gdG9vbHMvY3VzdG9tIHN5c3RlbSBwcm9tcHQ7IE9wdXM0LjhncmFkZXI7IEZhYmxlNS4xIHNhZmV0eSBmYWxsYmFjayB0b09wdXM1Lg","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"Raw rubric score; adaptive max; five trials; no tools/custom system prompt; Opus4.8grader; Fable5.1 safety fallback toOpus5."},{"benchmarkId":"healthbench-professional-percent","id":"healthbench-professional-percent:anthropic-fable-5-1-card:UmF3IHJ1YnJpYyBzY29yZTsgYWRhcHRpdmUgbWF4OyBmaXZlIHRyaWFsczsgbm8gdG9vbHMvY3VzdG9tIHN5c3RlbSBwcm9tcHQ7IE9wdXM0LjhncmFkZXI7IEZhYmxlNS4xIHNhZmV0eSBmYWxsYmFjayB0b09wdXM1Lg","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"Raw rubric score; adaptive max; five trials; no tools/custom system prompt; Opus4.8grader; Fable5.1 safety fallback toOpus5."},{"benchmarkId":"healthbench","id":"healthbench:anthropic-fable-5-1-card:TGVuZ3RoLWFkanVzdGVkIHNjb3JlIHVzaW5nIEdQVDUuNWNhcmQgbWV0aG9kOyBvdGhlcndpc2UgcmF3IGV2YWx1YXRpb24gY29uZmlndXJhdGlvbi4","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"Length-adjusted score using GPT5.5card method; otherwise raw evaluation configuration."},{"benchmarkId":"healthbench-professional-percent","id":"healthbench-professional-percent:anthropic-fable-5-1-card:TGVuZ3RoLWFkanVzdGVkIHNjb3JlOyBIZWFsdGhCZW5jaCBQcm9mZXNzaW9uYWwgcGFwZXIgbWV0aG9kOyBubyB0b29sczsgT3B1czQuOGdyYWRlcjsgZml2ZSB0cmlhbHMu","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"Length-adjusted score; HealthBench Professional paper method; no tools; Opus4.8grader; five trials."},{"benchmarkId":"gmmlu","id":"gmmlu:anthropic-fable-5-1-card:TWVhbiBhY2N1cmFjeTQybGFuZ3VhZ2VzOyBhZGFwdGl2ZSBtYXg7IG9uZSB0cmlhbDsgbm8gdG9vbHMvY3VzdG9tIHN5c3RlbSBwcm9tcHRzLg","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"Mean accuracy42languages; adaptive max; one trial; no tools/custom system prompts."},{"benchmarkId":"milu","id":"milu:anthropic-fable-5-1-card:TWVhbiBhY2N1cmFjeTExbGFuZ3VhZ2VzOyBhZGFwdGl2ZSBtYXg7IGZpdmUgdHJpYWxzOyBubyB0b29scy9jdXN0b20gc3lzdGVtIHByb21wdHMu","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"Mean accuracy11languages; adaptive max; five trials; no tools/custom system prompts."},{"benchmarkId":"biomysterybench-human-solvable","id":"biomysterybench-human-solvable:anthropic-fable-5-1-card:QW50aHJvcGljIGxpZmUtc2NpZW5jZXMgZXZhbHVhdGlvbjsgYmFzaC9lZGl0b3IvcGFja2FnZXMgZXhjZXB0IHByb3RlaW4gZGVzaWduIG5vIHRvb2xzOyBwcm90b2NvbHMgdXNlIGJhc2gvZWRpdG9yL3NlYXJjaDsgVW5kZXJzdGFuZGluZyByZXZpc2VkIG5ldHdvcmsgcmVzdHJpY3Rpb24u","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"Anthropic life-sciences evaluation; bash/editor/packages except protein design no tools; protocols use bash/editor/search; Understanding revised network restriction."},{"benchmarkId":"biomysterybench-human-difficult","id":"biomysterybench-human-difficult:anthropic-fable-5-1-card:QW50aHJvcGljIGxpZmUtc2NpZW5jZXMgZXZhbHVhdGlvbjsgYmFzaC9lZGl0b3IvcGFja2FnZXMgZXhjZXB0IHByb3RlaW4gZGVzaWduIG5vIHRvb2xzOyBwcm90b2NvbHMgdXNlIGJhc2gvZWRpdG9yL3NlYXJjaDsgVW5kZXJzdGFuZGluZyByZXZpc2VkIG5ldHdvcmsgcmVzdHJpY3Rpb24u","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"Anthropic life-sciences evaluation; bash/editor/packages except protein design no tools; protocols use bash/editor/search; Understanding revised network restriction."},{"benchmarkId":"spatialbench-verified","id":"spatialbench-verified:anthropic-fable-5-1-card:QW50aHJvcGljIGxpZmUtc2NpZW5jZXMgZXZhbHVhdGlvbjsgYmFzaC9lZGl0b3IvcGFja2FnZXMgZXhjZXB0IHByb3RlaW4gZGVzaWduIG5vIHRvb2xzOyBwcm90b2NvbHMgdXNlIGJhc2gvZWRpdG9yL3NlYXJjaDsgVW5kZXJzdGFuZGluZyByZXZpc2VkIG5ldHdvcmsgcmVzdHJpY3Rpb24u","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"Anthropic life-sciences evaluation; bash/editor/packages except protein design no tools; protocols use bash/editor/search; Understanding revised network restriction."},{"benchmarkId":"singlecellbench","id":"singlecellbench:anthropic-fable-5-1-card:QW50aHJvcGljIGxpZmUtc2NpZW5jZXMgZXZhbHVhdGlvbjsgYmFzaC9lZGl0b3IvcGFja2FnZXMgZXhjZXB0IHByb3RlaW4gZGVzaWduIG5vIHRvb2xzOyBwcm90b2NvbHMgdXNlIGJhc2gvZWRpdG9yL3NlYXJjaDsgVW5kZXJzdGFuZGluZyByZXZpc2VkIG5ldHdvcmsgcmVzdHJpY3Rpb24u","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"Anthropic life-sciences evaluation; bash/editor/packages except protein design no tools; protocols use bash/editor/search; Understanding revised network restriction."},{"benchmarkId":"proteingym-hard","id":"proteingym-hard:anthropic-fable-5-1-card:QW50aHJvcGljIGxpZmUtc2NpZW5jZXMgZXZhbHVhdGlvbjsgYmFzaC9lZGl0b3IvcGFja2FnZXMgZXhjZXB0IHByb3RlaW4gZGVzaWduIG5vIHRvb2xzOyBwcm90b2NvbHMgdXNlIGJhc2gvZWRpdG9yL3NlYXJjaDsgVW5kZXJzdGFuZGluZyByZXZpc2VkIG5ldHdvcmsgcmVzdHJpY3Rpb24u","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"Anthropic life-sciences evaluation; bash/editor/packages except protein design no tools; protocols use bash/editor/search; Understanding revised network restriction."},{"benchmarkId":"protein-design-sequence-generation-revised-grader","id":"protein-design-sequence-generation-revised-grader:anthropic-fable-5-1-card:QW50aHJvcGljIGxpZmUtc2NpZW5jZXMgZXZhbHVhdGlvbjsgYmFzaC9lZGl0b3IvcGFja2FnZXMgZXhjZXB0IHByb3RlaW4gZGVzaWduIG5vIHRvb2xzOyBwcm90b2NvbHMgdXNlIGJhc2gvZWRpdG9yL3NlYXJjaDsgVW5kZXJzdGFuZGluZyByZXZpc2VkIG5ldHdvcmsgcmVzdHJpY3Rpb24u","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"Anthropic life-sciences evaluation; bash/editor/packages except protein design no tools; protocols use bash/editor/search; Understanding revised network restriction."},{"benchmarkId":"protein-design-library-ranking","id":"protein-design-library-ranking:anthropic-fable-5-1-card:QW50aHJvcGljIGxpZmUtc2NpZW5jZXMgZXZhbHVhdGlvbjsgYmFzaC9lZGl0b3IvcGFja2FnZXMgZXhjZXB0IHByb3RlaW4gZGVzaWduIG5vIHRvb2xzOyBwcm90b2NvbHMgdXNlIGJhc2gvZWRpdG9yL3NlYXJjaDsgVW5kZXJzdGFuZGluZyByZXZpc2VkIG5ldHdvcmsgcmVzdHJpY3Rpb24u","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"Anthropic life-sciences evaluation; bash/editor/packages except protein design no tools; protocols use bash/editor/search; Understanding revised network restriction."},{"benchmarkId":"organic-chemistry-2-revised","id":"organic-chemistry-2-revised:anthropic-fable-5-1-card:QW50aHJvcGljIGxpZmUtc2NpZW5jZXMgZXZhbHVhdGlvbjsgYmFzaC9lZGl0b3IvcGFja2FnZXMgZXhjZXB0IHByb3RlaW4gZGVzaWduIG5vIHRvb2xzOyBwcm90b2NvbHMgdXNlIGJhc2gvZWRpdG9yL3NlYXJjaDsgVW5kZXJzdGFuZGluZyByZXZpc2VkIG5ldHdvcmsgcmVzdHJpY3Rpb24u","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"Anthropic life-sciences evaluation; bash/editor/packages except protein design no tools; protocols use bash/editor/search; Understanding revised network restriction."},{"benchmarkId":"protocols-troubleshooting","id":"protocols-troubleshooting:anthropic-fable-5-1-card:QW50aHJvcGljIGxpZmUtc2NpZW5jZXMgZXZhbHVhdGlvbjsgYmFzaC9lZGl0b3IvcGFja2FnZXMgZXhjZXB0IHByb3RlaW4gZGVzaWduIG5vIHRvb2xzOyBwcm90b2NvbHMgdXNlIGJhc2gvZWRpdG9yL3NlYXJjaDsgVW5kZXJzdGFuZGluZyByZXZpc2VkIG5ldHdvcmsgcmVzdHJpY3Rpb24u","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"Anthropic life-sciences evaluation; bash/editor/packages except protein design no tools; protocols use bash/editor/search; Understanding revised network restriction."},{"benchmarkId":"protocols-understanding-network-restricted","id":"protocols-understanding-network-restricted:anthropic-fable-5-1-card:QW50aHJvcGljIGxpZmUtc2NpZW5jZXMgZXZhbHVhdGlvbjsgYmFzaC9lZGl0b3IvcGFja2FnZXMgZXhjZXB0IHByb3RlaW4gZGVzaWduIG5vIHRvb2xzOyBwcm90b2NvbHMgdXNlIGJhc2gvZWRpdG9yL3NlYXJjaDsgVW5kZXJzdGFuZGluZyByZXZpc2VkIG5ldHdvcmsgcmVzdHJpY3Rpb24u","name":"Claude Fable 5.1 and Claude Mythos 5.1 System Card","notes":"Anthropic life-sciences evaluation; bash/editor/packages except protein design no tools; protocols use bash/editor/search; Understanding revised network restriction."},{"benchmarkId":"agents-last-exam-not-specified","id":"agents-last-exam-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","name":"GPT-6 Astra: A new generation of intelligence","notes":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"benchmarkId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline","id":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IG9mZmxpbmUgc2V0OyBwYXJ0aWFsIGNyZWRpdDsgdjIwMjYuMDguMDg7IG9mZmljaWFsIHRhc2svZ3JhZGluZyBzZXR0aW5ncw","name":"GPT-6 Astra: A new generation of intelligence","notes":"Maximum reported across reasoning efforts; OpenAI research environment or API; offline set; partial credit; v2026.08.08; official task/grading settings"},{"benchmarkId":"screenspot-pro-no-tools-not-specified","id":"screenspot-pro-no-tools-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IG5vIHRvb2xz","name":"GPT-6 Astra: A new generation of intelligence","notes":"Maximum reported across reasoning efforts; OpenAI research environment or API; no tools"},{"benchmarkId":"automationbench-not-specified","id":"automationbench-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","name":"GPT-6 Astra: A new generation of intelligence","notes":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"benchmarkId":"benchcad-not-specified","id":"benchcad-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHRvb2xzIGVuYWJsZWQ","name":"GPT-6 Astra: A new generation of intelligence","notes":"Maximum reported across reasoning efforts; OpenAI research environment or API; tools enabled"},{"benchmarkId":"benchcad-not-specified","id":"benchcad-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHRvb2xzIGVuYWJsZWQ7IHRocmVlIEFudGhyb3BpYyBldmFsdWF0aW9uIG1vZGlmaWNhdGlvbnM","name":"GPT-6 Astra: A new generation of intelligence","notes":"Maximum reported across reasoning efforts; OpenAI research environment or API; tools enabled; three Anthropic evaluation modifications"},{"benchmarkId":"browsecomp-not-specified","id":"browsecomp-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","name":"GPT-6 Astra: A new generation of intelligence","notes":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"benchmarkId":"openscore-string-quartets-1-omr-ned-not-specified","id":"openscore-string-quartets-1-omr-ned-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","name":"GPT-6 Astra: A new generation of intelligence","notes":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"benchmarkId":"internal-design-tasks-not-specified","id":"internal-design-tasks-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","name":"GPT-6 Astra: A new generation of intelligence","notes":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"benchmarkId":"internal-data-science-tasks-not-specified","id":"internal-data-science-tasks-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","name":"GPT-6 Astra: A new generation of intelligence","notes":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"benchmarkId":"artificial-analysis-intelligence-index-v4-1-1-4-1-1","id":"artificial-analysis-intelligence-index-v4-1-1-4-1-1:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","name":"GPT-6 Astra: A new generation of intelligence","notes":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"benchmarkId":"terminal-bench-4-0","id":"terminal-bench-4-0:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","name":"GPT-6 Astra: A new generation of intelligence","notes":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"benchmarkId":"deepswe-v1-1-1-1-percent","id":"deepswe-v1-1-1-1-percent:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","name":"GPT-6 Astra: A new generation of intelligence","notes":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"benchmarkId":"frontiercode-1-1-extended-score-1-1","id":"frontiercode-1-1-extended-score-1-1:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IENvZGV4LXN0eWxlIGRldmVsb3BlciBpbnN0cnVjdGlvbiBvbiB0ZXN0cywgcmV1c2UgYW5kIHJlcG9zaXRvcnkgY29udmVudGlvbnM","name":"GPT-6 Astra: A new generation of intelligence","notes":"Maximum reported across reasoning efforts; OpenAI research environment or API; Codex-style developer instruction on tests, reuse and repository conventions"},{"benchmarkId":"frontiercode-1-1-extended-score-1-1","id":"frontiercode-1-1-extended-score-1-1:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","name":"GPT-6 Astra: A new generation of intelligence","notes":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"benchmarkId":"frontiercode-1-1-main-score-1-1","id":"frontiercode-1-1-main-score-1-1:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IENvZGV4LXN0eWxlIGRldmVsb3BlciBpbnN0cnVjdGlvbiBvbiB0ZXN0cywgcmV1c2UgYW5kIHJlcG9zaXRvcnkgY29udmVudGlvbnM","name":"GPT-6 Astra: A new generation of intelligence","notes":"Maximum reported across reasoning efforts; OpenAI research environment or API; Codex-style developer instruction on tests, reuse and repository conventions"},{"benchmarkId":"frontiercode-1-1-main-score-1-1","id":"frontiercode-1-1-main-score-1-1:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","name":"GPT-6 Astra: A new generation of intelligence","notes":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"benchmarkId":"internal-database-migration-tasks-not-specified","id":"internal-database-migration-tasks-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","name":"GPT-6 Astra: A new generation of intelligence","notes":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"benchmarkId":"artificial-analysis-coding-agent-index-v1-4-1-4","id":"artificial-analysis-coding-agent-index-v1-4-1-4:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","name":"GPT-6 Astra: A new generation of intelligence","notes":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"benchmarkId":"terminal-bench-science-0-1-not-specified","id":"terminal-bench-science-0-1-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","name":"GPT-6 Astra: A new generation of intelligence","notes":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"benchmarkId":"frontiermath-tier-4-2","id":"frontiermath-tier-4-2:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","name":"GPT-6 Astra: A new generation of intelligence","notes":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"benchmarkId":"gpqa-diamond-not-specified","id":"gpqa-diamond-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","name":"GPT-6 Astra: A new generation of intelligence","notes":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"benchmarkId":"humanity-s-last-exam-w-tools-not-specified","id":"humanity-s-last-exam-w-tools-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHRvb2xzIGVuYWJsZWQ","name":"GPT-6 Astra: A new generation of intelligence","notes":"Maximum reported across reasoning efforts; OpenAI research environment or API; tools enabled"},{"benchmarkId":"genebench-pro-13","id":"genebench-pro-13:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","name":"GPT-6 Astra: A new generation of intelligence","notes":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"benchmarkId":"medchembench-internal-not-specified","id":"medchembench-internal-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","name":"GPT-6 Astra: A new generation of intelligence","notes":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"benchmarkId":"lifescibench-gold-v1","id":"lifescibench-gold-v1:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","name":"GPT-6 Astra: A new generation of intelligence","notes":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"benchmarkId":"healthbench-professional-length-adjusted-not-specified","id":"healthbench-professional-length-adjusted-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IG9mZmljaWFsIHBhcGVyIHNjb3Jpbmc7IGxlbmd0aC1hZGp1c3RlZA","name":"GPT-6 Astra: A new generation of intelligence","notes":"Maximum reported across reasoning efforts; OpenAI research environment or API; official paper scoring; length-adjusted"},{"benchmarkId":"healthbench-professional-length-adjusted-not-specified","id":"healthbench-professional-length-adjusted-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IG9mZmljaWFsIHBhcGVyIHNjb3Jpbmc7IGxlbmd0aC1hZGp1c3RlZDsgT3BlbkFJIHJlcHJvZHVjdGlvbjsgR1BULTUuNCBncmFkZXI7IHVuY2xpcHBlZDsgT3B1czUgZmFsbGJhY2sgZm9yIHByb3ZpZGVyIHJlZnVzYWxz","name":"GPT-6 Astra: A new generation of intelligence","notes":"Maximum reported across reasoning efforts; OpenAI research environment or API; official paper scoring; length-adjusted; OpenAI reproduction; GPT-5.4 grader; unclipped; Opus5 fallback for provider refusals"},{"benchmarkId":"healthbench-professional-length-adjusted-not-specified","id":"healthbench-professional-length-adjusted-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IG9mZmljaWFsIHBhcGVyIHNjb3Jpbmc7IGxlbmd0aC1hZGp1c3RlZDsgT3BlbkFJIHJlcHJvZHVjdGlvbjsgR1BULTUuNCBncmFkZXI7IHVuY2xpcHBlZA","name":"GPT-6 Astra: A new generation of intelligence","notes":"Maximum reported across reasoning efforts; OpenAI research environment or API; official paper scoring; length-adjusted; OpenAI reproduction; GPT-5.4 grader; unclipped"},{"benchmarkId":"exploitbench-not-specified","id":"exploitbench-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHJlZHVjZWQgb3IgYWJzZW50IHByb2R1Y3Rpb24gc2FmZWd1YXJkcw","name":"GPT-6 Astra: A new generation of intelligence","notes":"Maximum reported across reasoning efforts; OpenAI research environment or API; reduced or absent production safeguards"},{"benchmarkId":"exploitgym-not-specified","id":"exploitgym-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHJlZHVjZWQgb3IgYWJzZW50IHByb2R1Y3Rpb24gc2FmZWd1YXJkczsgdjEgb2ZmbGluZSBlbnZpcm9ubWVudDsgbm8gcnVudGltZSBwYWNrYWdlIGluc3RhbGxhdGlvbjsgdG9rZW4gY2FwcGVkLCBubyB3YWxsLWNsb2NrIGNhcA","name":"GPT-6 Astra: A new generation of intelligence","notes":"Maximum reported across reasoning efforts; OpenAI research environment or API; reduced or absent production safeguards; v1 offline environment; no runtime package installation; token capped, no wall-clock cap"},{"benchmarkId":"exploitgym-not-specified","id":"exploitgym-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHJlZHVjZWQgb3IgYWJzZW50IHByb2R1Y3Rpb24gc2FmZWd1YXJkczsgbGF1bmNoLXJlcG9ydGVkIGNvbmZpZ3VyYXRpb24","name":"GPT-6 Astra: A new generation of intelligence","notes":"Maximum reported across reasoning efforts; OpenAI research environment or API; reduced or absent production safeguards; launch-reported configuration"},{"benchmarkId":"exploitbench-june-aug-2026-june-aug2026","id":"exploitbench-june-aug-2026-june-aug2026:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHJlZHVjZWQgb3IgYWJzZW50IHByb2R1Y3Rpb24gc2FmZWd1YXJkczsgMjAgdnVsbmVyYWJpbGl0aWVzIC8gMTMgQ2hyb21lIHJlbGVhc2Vz","name":"GPT-6 Astra: A new generation of intelligence","notes":"Maximum reported across reasoning efforts; OpenAI research environment or API; reduced or absent production safeguards; 20 vulnerabilities / 13 Chrome releases"},{"benchmarkId":"exploitbench-june-aug-2026-june-aug2026","id":"exploitbench-june-aug-2026-june-aug2026:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHJlZHVjZWQgb3IgYWJzZW50IHByb2R1Y3Rpb24gc2FmZWd1YXJkczsgMjAgdnVsbmVyYWJpbGl0aWVzIC8gMTMgQ2hyb21lIHJlbGVhc2VzOyAzMDAtdHVybiBsaW1pdA","name":"GPT-6 Astra: A new generation of intelligence","notes":"Maximum reported across reasoning efforts; OpenAI research environment or API; reduced or absent production safeguards; 20 vulnerabilities / 13 Chrome releases; 300-turn limit"},{"benchmarkId":"sre-bench-262-binaries-19-programs","id":"sre-bench-262-binaries-19-programs:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHJlZHVjZWQgb3IgYWJzZW50IHByb2R1Y3Rpb24gc2FmZWd1YXJkczsgcGFzc0AxOyBhbGwgc2l4IG9iamVjdGl2ZXMgcmVxdWlyZWQ","name":"GPT-6 Astra: A new generation of intelligence","notes":"Maximum reported across reasoning efforts; OpenAI research environment or API; reduced or absent production safeguards; pass@1; all six objectives required"},{"benchmarkId":"sec-bench-pro-may2026-revised-root-cause-grader","id":"sec-bench-pro-may2026-revised-root-cause-grader:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHJlZHVjZWQgb3IgYWJzZW50IHByb2R1Y3Rpb24gc2FmZWd1YXJkczsgTWF5MjAyNiBKYXZhU2NyaXB0IHN1YnNldCwxODMgdnVsbmVyYWJpbGl0aWVzOyByZXZpc2VkIGFnZW50IHJvb3QtY2F1c2UgZ3JhZGVy","name":"GPT-6 Astra: A new generation of intelligence","notes":"Maximum reported across reasoning efforts; OpenAI research environment or API; reduced or absent production safeguards; May2026 JavaScript subset,183 vulnerabilities; revised agent root-cause grader"},{"benchmarkId":"openai-mrcr-v2-8-needle-256k-512k-2","id":"openai-mrcr-v2-8-needle-256k-512k-2:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IDgtbmVlZGxlIDI1NkstNTEySw","name":"GPT-6 Astra: A new generation of intelligence","notes":"Maximum reported across reasoning efforts; OpenAI research environment or API; 8-needle 256K-512K"},{"benchmarkId":"openai-mrcr-v2-8-needle-512k-1m-2","id":"openai-mrcr-v2-8-needle-512k-1m-2:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IDgtbmVlZGxlIDUxMkstMU0","name":"GPT-6 Astra: A new generation of intelligence","notes":"Maximum reported across reasoning efforts; OpenAI research environment or API; 8-needle 512K-1M"},{"benchmarkId":"arc-agi-3","id":"arc-agi-3:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IFJlc3BvbnNlcyBBUEkgaGFybmVzcyB3aXRoIHR3byBkb2N1bWVudGVkIHNldHRpbmcgY2hhbmdlcw","name":"GPT-6 Astra: A new generation of intelligence","notes":"Maximum reported across reasoning efforts; OpenAI research environment or API; Responses API harness with two documented setting changes"},{"benchmarkId":"arc-agi-3","id":"arc-agi-3:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","name":"GPT-6 Astra: A new generation of intelligence","notes":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"benchmarkId":"arc-agi-2-percent","id":"arc-agi-2-percent:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","name":"GPT-6 Astra: A new generation of intelligence","notes":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"benchmarkId":"arc-agi-1","id":"arc-agi-1:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","name":"GPT-6 Astra: A new generation of intelligence","notes":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"benchmarkId":"agents-last-exam-not-specified","id":"agents-last-exam-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","name":"GPT-5.6: Frontier intelligence that scales with your ambition","notes":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"benchmarkId":"gdpval-aa-v2-2","id":"gdpval-aa-v2-2:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","name":"GPT-5.6: Frontier intelligence that scales with your ambition","notes":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"benchmarkId":"management-consulting-tasks-internal-not-specified","id":"management-consulting-tasks-internal-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","name":"GPT-5.6: Frontier intelligence that scales with your ambition","notes":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"benchmarkId":"big-finance-bench-not-specified","id":"big-finance-bench-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","name":"GPT-5.6: Frontier intelligence that scales with your ambition","notes":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"benchmarkId":"artificial-analysis-intelligence-index-v4-1-4-1","id":"artificial-analysis-intelligence-index-v4-1-4-1:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","name":"GPT-5.6: Frontier intelligence that scales with your ambition","notes":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"benchmarkId":"artificial-analysis-coding-agent-index-v1-1-1-1","id":"artificial-analysis-coding-agent-index-v1-1-1-1:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","name":"GPT-5.6: Frontier intelligence that scales with your ambition","notes":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"benchmarkId":"swe-bench-pro-not-specified","id":"swe-bench-pro-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","name":"GPT-5.6: Frontier intelligence that scales with your ambition","notes":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"benchmarkId":"deepswe-v1-1-1-1-percent","id":"deepswe-v1-1-1-1-percent:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","name":"GPT-5.6: Frontier intelligence that scales with your ambition","notes":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"benchmarkId":"terminal-bench-2-1-percent","id":"terminal-bench-2-1-percent:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","name":"GPT-5.6: Frontier intelligence that scales with your ambition","notes":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"benchmarkId":"terminal-bench-2-1-percent","id":"terminal-bench-2-1-percent:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IFVsdHJhLCBmb3VyLWFnZW50IG9yY2hlc3RyYXRpb24","name":"GPT-5.6: Frontier intelligence that scales with your ambition","notes":"Launch table reported configuration; per-cell reasoning effort unspecified; Ultra, four-agent orchestration"},{"benchmarkId":"genebench-pro-not-specified","id":"genebench-pro-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","name":"GPT-5.6: Frontier intelligence that scales with your ambition","notes":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"benchmarkId":"lifescibench-not-specified","id":"lifescibench-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","name":"GPT-5.6: Frontier intelligence that scales with your ambition","notes":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"benchmarkId":"medchembench-internal-not-specified","id":"medchembench-internal-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","name":"GPT-5.6: Frontier intelligence that scales with your ambition","notes":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"benchmarkId":"healthbench-professional-not-specified","id":"healthbench-professional-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IG9mZmljaWFsIHBhcGVyIHNjb3Jpbmc7IGxlbmd0aC1hZGp1c3RlZA","name":"GPT-5.6: Frontier intelligence that scales with your ambition","notes":"Launch table reported configuration; per-cell reasoning effort unspecified; official paper scoring; length-adjusted"},{"benchmarkId":"osworld-2-0-percent","id":"osworld-2-0-percent:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","name":"GPT-5.6: Frontier intelligence that scales with your ambition","notes":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"benchmarkId":"browsecomp-not-specified","id":"browsecomp-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","name":"GPT-5.6: Frontier intelligence that scales with your ambition","notes":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"benchmarkId":"browsecomp-not-specified","id":"browsecomp-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IFVsdHJhLCBmb3VyLWFnZW50IG9yY2hlc3RyYXRpb24","name":"GPT-5.6: Frontier intelligence that scales with your ambition","notes":"Launch table reported configuration; per-cell reasoning effort unspecified; Ultra, four-agent orchestration"},{"benchmarkId":"benchcad-not-specified","id":"benchcad-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","name":"GPT-5.6: Frontier intelligence that scales with your ambition","notes":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"benchmarkId":"benchcad-python-tool-not-specified","id":"benchcad-python-tool-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IFB5dGhvbiB0b29sIGVuYWJsZWQ","name":"GPT-5.6: Frontier intelligence that scales with your ambition","notes":"Launch table reported configuration; per-cell reasoning effort unspecified; Python tool enabled"},{"benchmarkId":"capture-the-flag-challenges-not-specified","id":"capture-the-flag-challenges-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IHJlZHVjZWQgb3IgYWJzZW50IHByb2R1Y3Rpb24gc2FmZWd1YXJkcw","name":"GPT-5.6: Frontier intelligence that scales with your ambition","notes":"Launch table reported configuration; per-cell reasoning effort unspecified; reduced or absent production safeguards"},{"benchmarkId":"sec-bench-pro-may2026-public-grader","id":"sec-bench-pro-may2026-public-grader:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IHJlZHVjZWQgb3IgYWJzZW50IHByb2R1Y3Rpb24gc2FmZWd1YXJkczsgcHVibGljIGdyYWRlcjsgTWF5MjAyNiBKYXZhU2NyaXB0IHN1YnNldA","name":"GPT-5.6: Frontier intelligence that scales with your ambition","notes":"Launch table reported configuration; per-cell reasoning effort unspecified; reduced or absent production safeguards; public grader; May2026 JavaScript subset"},{"benchmarkId":"sec-bench-pro-may2026-public-grader","id":"sec-bench-pro-may2026-public-grader:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IFVsdHJhLCBmb3VyLWFnZW50IG9yY2hlc3RyYXRpb247IHJlZHVjZWQgb3IgYWJzZW50IHByb2R1Y3Rpb24gc2FmZWd1YXJkczsgcHVibGljIGdyYWRlcjsgTWF5MjAyNiBKYXZhU2NyaXB0IHN1YnNldA","name":"GPT-5.6: Frontier intelligence that scales with your ambition","notes":"Launch table reported configuration; per-cell reasoning effort unspecified; Ultra, four-agent orchestration; reduced or absent production safeguards; public grader; May2026 JavaScript subset"},{"benchmarkId":"exploitbench-not-specified","id":"exploitbench-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IHJlZHVjZWQgb3IgYWJzZW50IHByb2R1Y3Rpb24gc2FmZWd1YXJkczsgRXhwbG9pdEJlbmNoIEFQSSBoYXJuZXNzOyBmaXZlIHNlZWRzOyByZWFzb25pbmcgY29udGludWl0eQ","name":"GPT-5.6: Frontier intelligence that scales with your ambition","notes":"Launch table reported configuration; per-cell reasoning effort unspecified; reduced or absent production safeguards; ExploitBench API harness; five seeds; reasoning continuity"},{"benchmarkId":"exploitgym-not-specified","id":"exploitgym-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IHJlZHVjZWQgb3IgYWJzZW50IHByb2R1Y3Rpb24gc2FmZWd1YXJkczsgc2l4LWhvdXIgZXZhbHVhdGlvbiBjYXA7IGFscGhhIEFQSSBsYXRlbmN5IHJlc2NhbGVkIHRvIHB1YmxpYyBBUEk","name":"GPT-5.6: Frontier intelligence that scales with your ambition","notes":"Launch table reported configuration; per-cell reasoning effort unspecified; reduced or absent production safeguards; six-hour evaluation cap; alpha API latency rescaled to public API"},{"benchmarkId":"internal-research-debugging-evaluation-not-specified","id":"internal-research-debugging-evaluation-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","name":"GPT-5.6: Frontier intelligence that scales with your ambition","notes":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"benchmarkId":"kernelgen-1p-not-specified","id":"kernelgen-1p-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","name":"GPT-5.6: Frontier intelligence that scales with your ambition","notes":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"benchmarkId":"nanogpt-not-specified","id":"nanogpt-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","name":"GPT-5.6: Frontier intelligence that scales with your ambition","notes":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"benchmarkId":"posttrainbench-lite-not-specified","id":"posttrainbench-lite-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","name":"GPT-5.6: Frontier intelligence that scales with your ambition","notes":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"benchmarkId":"rsi-index-not-specified","id":"rsi-index-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","name":"GPT-5.6: Frontier intelligence that scales with your ambition","notes":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"benchmarkId":"mmmu-pro-no-tools-not-specified","id":"mmmu-pro-no-tools-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IG5vIHRvb2xz","name":"GPT-5.6: Frontier intelligence that scales with your ambition","notes":"Launch table reported configuration; per-cell reasoning effort unspecified; no tools"},{"benchmarkId":"mmmu-pro-with-tools-not-specified","id":"mmmu-pro-with-tools-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IHRvb2xzIGVuYWJsZWQ","name":"GPT-5.6: Frontier intelligence that scales with your ambition","notes":"Launch table reported configuration; per-cell reasoning effort unspecified; tools enabled"},{"benchmarkId":"gdp-pdf-not-specified","id":"gdp-pdf-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","name":"GPT-5.6: Frontier intelligence that scales with your ambition","notes":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"benchmarkId":"gpqa-diamond-not-specified","id":"gpqa-diamond-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","name":"GPT-5.6: Frontier intelligence that scales with your ambition","notes":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"benchmarkId":"frontiermath-tier-1-3-2","id":"frontiermath-tier-1-3-2:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","name":"GPT-5.6: Frontier intelligence that scales with your ambition","notes":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"benchmarkId":"frontiermath-tier-4-2","id":"frontiermath-tier-4-2:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","name":"GPT-5.6: Frontier intelligence that scales with your ambition","notes":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"benchmarkId":"automationbench-not-specified","id":"automationbench-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","name":"GPT-5.6: Frontier intelligence that scales with your ambition","notes":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"benchmarkId":"toolathlon-not-specified","id":"toolathlon-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","name":"GPT-5.6: Frontier intelligence that scales with your ambition","notes":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"benchmarkId":"openai-mrcr-v2-8-needle-256k-512k-2","id":"openai-mrcr-v2-8-needle-256k-512k-2:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IDgtbmVlZGxlIDI1NkstNTEySw","name":"GPT-5.6: Frontier intelligence that scales with your ambition","notes":"Launch table reported configuration; per-cell reasoning effort unspecified; 8-needle 256K-512K"},{"benchmarkId":"openai-mrcr-v2-8-needle-512k-1m-2","id":"openai-mrcr-v2-8-needle-512k-1m-2:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IDgtbmVlZGxlIDUxMkstMU0","name":"GPT-5.6: Frontier intelligence that scales with your ambition","notes":"Launch table reported configuration; per-cell reasoning effort unspecified; 8-needle 512K-1M"},{"benchmarkId":"graphwalks-bfs-256k-f1-not-specified","id":"graphwalks-bfs-256k-f1-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IEJGUyAyNTZrIGYx","name":"GPT-5.6: Frontier intelligence that scales with your ambition","notes":"Launch table reported configuration; per-cell reasoning effort unspecified; BFS 256k f1"},{"benchmarkId":"graphwalks-bfs-1mil-f1-not-specified","id":"graphwalks-bfs-1mil-f1-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IEJGUyAxbWlsIGYx","name":"GPT-5.6: Frontier intelligence that scales with your ambition","notes":"Launch table reported configuration; per-cell reasoning effort unspecified; BFS 1mil f1"},{"benchmarkId":"arc-agi-3","id":"arc-agi-3:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","name":"GPT-5.6: Frontier intelligence that scales with your ambition","notes":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"benchmarkId":"arc-agi-3","id":"arc-agi-3:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IGhpZ2ggcmVhc29uaW5nLCBub3QgbWF4","name":"GPT-5.6: Frontier intelligence that scales with your ambition","notes":"Launch table reported configuration; per-cell reasoning effort unspecified; high reasoning, not max"},{"benchmarkId":"healthbench-professional-not-specified-score-0-100","id":"healthbench-professional-not-specified-score-0-100:openai-astra-system-card:bGVuZ3RoLWFkanVzdGVkOyBvZmZpY2lhbCBIZWFsdGhCZW5jaCBzY29yaW5n","name":"GPT-6 Astra System Card","notes":"length-adjusted; official HealthBench scoring"},{"benchmarkId":"healthbench-professional-not-specified-score-0-100","id":"healthbench-professional-not-specified-score-0-100:openai-astra-system-card:dW5hZGp1c3RlZDsgb2ZmaWNpYWwgSGVhbHRoQmVuY2ggc2NvcmluZw","name":"GPT-6 Astra System Card","notes":"unadjusted; official HealthBench scoring"},{"benchmarkId":"healthbench-not-specified","id":"healthbench-not-specified:openai-astra-system-card:bGVuZ3RoLWFkanVzdGVkOyBvZmZpY2lhbCBIZWFsdGhCZW5jaCBzY29yaW5n","name":"GPT-6 Astra System Card","notes":"length-adjusted; official HealthBench scoring"},{"benchmarkId":"healthbench-not-specified","id":"healthbench-not-specified:openai-astra-system-card:dW5hZGp1c3RlZDsgb2ZmaWNpYWwgSGVhbHRoQmVuY2ggc2NvcmluZw","name":"GPT-6 Astra System Card","notes":"unadjusted; official HealthBench scoring"},{"benchmarkId":"healthbench-hard-not-specified","id":"healthbench-hard-not-specified:openai-astra-system-card:bGVuZ3RoLWFkanVzdGVkOyBvZmZpY2lhbCBIZWFsdGhCZW5jaCBzY29yaW5n","name":"GPT-6 Astra System Card","notes":"length-adjusted; official HealthBench scoring"},{"benchmarkId":"healthbench-hard-not-specified","id":"healthbench-hard-not-specified:openai-astra-system-card:dW5hZGp1c3RlZDsgb2ZmaWNpYWwgSGVhbHRoQmVuY2ggc2NvcmluZw","name":"GPT-6 Astra System Card","notes":"unadjusted; official HealthBench scoring"},{"benchmarkId":"healthbench-consensus-not-specified","id":"healthbench-consensus-not-specified:openai-astra-system-card:bGVuZ3RoLWFkanVzdGVkOyBvZmZpY2lhbCBIZWFsdGhCZW5jaCBzY29yaW5n","name":"GPT-6 Astra System Card","notes":"length-adjusted; official HealthBench scoring"},{"benchmarkId":"healthbench-consensus-not-specified","id":"healthbench-consensus-not-specified:openai-astra-system-card:dW5hZGp1c3RlZDsgb2ZmaWNpYWwgSGVhbHRoQmVuY2ggc2NvcmluZw","name":"GPT-6 Astra System Card","notes":"unadjusted; official HealthBench scoring"},{"benchmarkId":"internal-research-debugging-evaluation-41-research-bugs-6-alignment-auditing-tasks","id":"internal-research-debugging-evaluation-41-research-bugs-6-alignment-auditing-tasks:openai-astra-system-card:U3lzdGVtLWNhcmQgcmVwb3J0ZWQgY29uZmlndXJhdGlvbjsgcmVhc29uaW5nIGVmZm9ydCB1bnNwZWNpZmllZA","name":"GPT-6 Astra System Card","notes":"System-card reported configuration; reasoning effort unspecified"},{"benchmarkId":"sre-bench-262-binaries-19-programs","id":"sre-bench-262-binaries-19-programs:openai-astra-system-card:cGFzc0A0OyBmb3VyIGluZGVwZW5kZW50IHRyaWFsczsgYWxsIHNpeCBvYmplY3RpdmVzIHJlcXVpcmVkOyByZWR1Y2VkIHByb2R1Y3Rpb24gc2FmZWd1YXJkcw","name":"GPT-6 Astra System Card","notes":"pass@4; four independent trials; all six objectives required; reduced production safeguards"},{"benchmarkId":"sandbox-bench-september2026-internal","id":"sandbox-bench-september2026-internal:openai-astra-system-card:MjIgaXNvbGF0ZWQgQ1RGLXN0eWxlIHRhcmdldHM7IHByb3RlY3RlZC1mbGFnIHN1Y2Nlc3MgbWV0cmlj","name":"GPT-6 Astra System Card","notes":"22 isolated CTF-style targets; protected-flag success metric"},{"benchmarkId":"no-cot-math-time-horizon-not-specified","id":"no-cot-math-time-horizon-not-specified:openai-astra-system-card:VUsgQUlTSTsgc2luZ2xlIGZvcndhcmQgcGFzczsgbm8gY2hhaW4gb2YgdGhvdWdodA","name":"GPT-6 Astra System Card","notes":"UK AISI; single forward pass; no chain of thought"},{"benchmarkId":"protocolqa-open-ended-108-questions","id":"protocolqa-open-ended-108-questions:openai-astra-system-card:b2JzZXJ2ZWQgcGVyZm9ybWFuY2U","name":"GPT-6 Astra System Card","notes":"observed performance"},{"benchmarkId":"protocolqa-open-ended-108-questions","id":"protocolqa-open-ended-108-questions:openai-astra-system-card:cmVmdXNhbC1hZGp1c3RlZCB1cHBlciBlc3RpbWF0ZTsgcmVmdXNhbHMgY291bnRlZCBhcyBzdWNjZXNzZXM","name":"GPT-6 Astra System Card","notes":"refusal-adjusted upper estimate; refusals counted as successes"},{"benchmarkId":"tacit-knowledge-and-troubleshooting-60-questions","id":"tacit-knowledge-and-troubleshooting-60-questions:openai-astra-system-card:b2JzZXJ2ZWQgcGVyZm9ybWFuY2U","name":"GPT-6 Astra System Card","notes":"observed performance"},{"benchmarkId":"tacit-knowledge-and-troubleshooting-60-questions","id":"tacit-knowledge-and-troubleshooting-60-questions:openai-astra-system-card:cmVmdXNhbC1hZGp1c3RlZCB1cHBlciBlc3RpbWF0ZTsgcmVmdXNhbHMgY291bnRlZCBhcyBzdWNjZXNzZXM","name":"GPT-6 Astra System Card","notes":"refusal-adjusted upper estimate; refusals counted as successes"},{"benchmarkId":"troubleshootingbench-156-questions-52-protocols","id":"troubleshootingbench-156-questions-52-protocols:openai-astra-system-card:b2JzZXJ2ZWQgcGVyZm9ybWFuY2U","name":"GPT-6 Astra System Card","notes":"observed performance"},{"benchmarkId":"troubleshootingbench-156-questions-52-protocols","id":"troubleshootingbench-156-questions-52-protocols:openai-astra-system-card:cmVmdXNhbC1hZGp1c3RlZCB1cHBlciBlc3RpbWF0ZTsgcmVmdXNhbHMgY291bnRlZCBhcyBzdWNjZXNzZXM","name":"GPT-6 Astra System Card","notes":"refusal-adjusted upper estimate; refusals counted as successes"},{"benchmarkId":"shp2-protein-function-prediction-3-unpublished-assay-datasets","id":"shp2-protein-function-prediction-3-unpublished-assay-datasets:openai-astra-system-card:U3lzdGVtLWNhcmQgcmVwb3J0ZWQgY29uZmlndXJhdGlvbjsgcmVhc29uaW5nIGVmZm9ydCB1bnNwZWNpZmllZA","name":"GPT-6 Astra System Card","notes":"System-card reported configuration; reasoning effort unspecified"},{"benchmarkId":"coronavirus-ace2-cell-entry-screen-not-specified","id":"coronavirus-ace2-cell-entry-screen-not-specified:openai-astra-system-card:U3lzdGVtLWNhcmQgcmVwb3J0ZWQgY29uZmlndXJhdGlvbjsgcmVhc29uaW5nIGVmZm9ydCB1bnNwZWNpZmllZA","name":"GPT-6 Astra System Card","notes":"System-card reported configuration; reasoning effort unspecified"},{"benchmarkId":"phage-plasmid-co-evolution-not-specified","id":"phage-plasmid-co-evolution-not-specified:openai-astra-system-card:U3lzdGVtLWNhcmQgcmVwb3J0ZWQgY29uZmlndXJhdGlvbjsgcmVhc29uaW5nIGVmZm9ydCB1bnNwZWNpZmllZA","name":"GPT-6 Astra System Card","notes":"System-card reported configuration; reasoning effort unspecified"},{"benchmarkId":"terminal-bench-science-0-1","id":"terminal-bench-science-0-1:openai-astra-launch:TG93ZXItY29zdCBzZXR0aW5nOyBleGFjdCBlZmZvcnQgbm90IHNwZWNpZmllZA","name":"GPT-6 Astra: A new generation of intelligence","notes":"Lower-cost setting; exact effort not specified"},{"benchmarkId":"gpqa-diamond-not-specified","id":"gpqa-diamond-not-specified:openai-astra-launch:TG93ZXItY29zdCBzZXR0aW5nOyBleGFjdCBlZmZvcnQgbm90IHNwZWNpZmllZA","name":"GPT-6 Astra: A new generation of intelligence","notes":"Lower-cost setting; exact effort not specified"},{"benchmarkId":"exploitbench-june-aug-2026-june-aug2026","id":"exploitbench-june-aug-2026-june-aug2026:openai-astra-launch:U2ltaWxhciBzZXR0aW5ncyB3aXRoIGZld2VyMzAwLXR1cm4tbGltaXQgaW50ZXJydXB0aW9uczsgcHJvZHVjdGlvbiBzYWZlZ3VhcmRzIGFic2VudA","name":"GPT-6 Astra: A new generation of intelligence","notes":"Similar settings with fewer300-turn-limit interruptions; production safeguards absent"},{"benchmarkId":"exploitgym-not-specified","id":"exploitgym-not-specified:openai-sol-launch:VHdvLWhvdXIgY2FwOyBhbHBoYSBBUEkgbGF0ZW5jeSByZXNjYWxlZDsgcmVkdWNlZCBzYWZlZ3VhcmRz","name":"GPT-5.6: Frontier intelligence that scales with your ambition","notes":"Two-hour cap; alpha API latency rescaled; reduced safeguards"},{"benchmarkId":"terminal-bench-2-1","id":"terminal-bench-2-1:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gUXdlbiBDbGF1ZGUgQ29kZSBhdmdAMTAsNWggdGltZW91dCwxMzEwNzIgb3V0cHV0OyBjb21wYXJhdG9ycyBiZXN0IHB1Ymxpc2hlZCBhY3Jvc3MgaGFybmVzc2VzLg","name":"Qwen/Qwen3.8-2.4T-A95B","notes":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Qwen Claude Code avg@10,5h timeout,131072 output; comparators best published across harnesses."},{"benchmarkId":"swe-bench-pro-source-release-snapshot-version-not-specified","id":"swe-bench-pro-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gUmVmaW5lZCB0YXNrIHNldCB3aXRoIHByb2JsZW1hdGljIHRhc2tzIGNvcnJlY3RlZDsgYWxsIGJhc2VsaW5lcyByZXJ1biwgQ2xhdWRlIENvZGUsdGVtcDEsdG9wX3AuOTUsMjU2SyBjb250ZXh0Lg","name":"Qwen/Qwen3.8-2.4T-A95B","notes":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Refined task set with problematic tasks corrected; all baselines rerun, Claude Code,temp1,top_p.95,256K context."},{"benchmarkId":"deepswe-1-1-pct","id":"deepswe-1-1-pct:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gQmVzdCBvZiBDbGF1ZGUgQ29kZSBhbmQgbWluaS1TV0UtYWdlbnQ7IFF3ZW4gYmVzdCBDbGF1ZGUgQ29kZTsgdGVtcDEsdG9wX3AuOTUsMjU2Sy4","name":"Qwen/Qwen3.8-2.4T-A95B","notes":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Best of Claude Code and mini-SWE-agent; Qwen best Claude Code; temp1,top_p.95,256K."},{"benchmarkId":"nl2repo-bench-source-release-snapshot-version-not-specified","id":"nl2repo-bench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","name":"Qwen/Qwen3.8-2.4T-A95B","notes":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"benchmarkId":"frontierswe-source-release-snapshot-version-not-specified","id":"frontierswe-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","name":"Qwen/Qwen3.8-2.4T-A95B","notes":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"benchmarkId":"mls-bench-lite-source-release-snapshot-version-not-specified","id":"mls-bench-lite-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","name":"Qwen/Qwen3.8-2.4T-A95B","notes":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"benchmarkId":"paperbench-source-release-snapshot-version-not-specified","id":"paperbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gQmFzaWNBZ2VudCBDb2RlLURldjsgT3B1czQuNiBqdWRnZSwzcnVucywxMmggZWFjaC4","name":"Qwen/Qwen3.8-2.4T-A95B","notes":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. BasicAgent Code-Dev; Opus4.6 judge,3runs,12h each."},{"benchmarkId":"androidbench-source-release-snapshot-version-not-specified","id":"androidbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","name":"Qwen/Qwen3.8-2.4T-A95B","notes":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"benchmarkId":"qwenswebench-source-release-snapshot-version-not-specified","id":"qwenswebench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gSW50ZXJuYWwgc29mdHdhcmUgZW5naW5lZXJpbmcsQ2xhdWRlQ29kZSxhdmdAMyw4aCwzMjc2OCBvdXRwdXQsdGVtcDEsMjU2Sy4","name":"Qwen/Qwen3.8-2.4T-A95B","notes":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal software engineering,ClaudeCode,avg@3,8h,32768 output,temp1,256K."},{"benchmarkId":"qwenqoderbench-source-release-snapshot-version-not-specified","id":"qwenqoderbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gSW50ZXJuYWwgUW9kZXIgdGFza3MsQ2xhdWRlQ29kZSxhdmdANSw2aCwzMjc2OCBvdXRwdXQsdGVtcDEsMjU2Sy4","name":"Qwen/Qwen3.8-2.4T-A95B","notes":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal Qoder tasks,ClaudeCode,avg@5,6h,32768 output,temp1,256K."},{"benchmarkId":"qwenreactbench-source-release-snapshot-version-not-specified","id":"qwenreactbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gSW50ZXJuYWwgYmlsaW5ndWFsIEVOL0NOIFJlYWN0IGJlbmNobWFyayw3Y2F0ZWdvcmllcyxDbGF1ZGVDb2RlLHJlbmRlcittdWx0aW1vZGFsanVkZ2UsQlQvRWxvLg","name":"Qwen/Qwen3.8-2.4T-A95B","notes":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal bilingual EN/CN React benchmark,7categories,ClaudeCode,render+multimodaljudge,BT/Elo."},{"benchmarkId":"qwensvgbench-source-release-snapshot-version-not-specified","id":"qwensvgbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gSW50ZXJuYWwgYmlsaW5ndWFsIEVOL0NOIFNWRyBiZW5jaG1hcmsscmVuZGVyK211bHRpbW9kYWxqdWRnZSxCVC9FbG8u","name":"Qwen/Qwen3.8-2.4T-A95B","notes":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal bilingual EN/CN SVG benchmark,render+multimodaljudge,BT/Elo."},{"benchmarkId":"coworkbench-source-release-snapshot-version-not-specified","id":"coworkbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gSW50ZXJuYWwgcHJvZmVzc2lvbmFsIHdvcmsgdGFza3MgYWNyb3NzIHNjaWVuY2UsZmluYW5jZSxsYXcsbWVkaWNhbCxwcm9kdWN0aXZpdHku","name":"Qwen/Qwen3.8-2.4T-A95B","notes":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal professional work tasks across science,finance,law,medical,productivity."},{"benchmarkId":"workspacebench-source-release-snapshot-version-not-specified","id":"workspacebench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","name":"Qwen/Qwen3.8-2.4T-A95B","notes":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"benchmarkId":"jobbench-source-release-snapshot-version-not-specified","id":"jobbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","name":"Qwen/Qwen3.8-2.4T-A95B","notes":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"benchmarkId":"skillsbench-source-release-snapshot-version-not-specified","id":"skillsbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gdjEuMSBwdWJsaWM4N3Rhc2tzLDNydW5zOyBBbnRocm9waWMgQ2xhdWRlQ29kZSxPcGVuQUkgQ29kZXgsUXdlbiBPcGVuQ29kZS4","name":"Qwen/Qwen3.8-2.4T-A95B","notes":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. v1.1 public87tasks,3runs; Anthropic ClaudeCode,OpenAI Codex,Qwen OpenCode."},{"benchmarkId":"agents-last-exam-pass-score-source-release-snapshot-version-not-specified","id":"agents-last-exam-pass-score-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gUGFzcw","name":"Qwen/Qwen3.8-2.4T-A95B","notes":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Pass"},{"benchmarkId":"agents-last-exam-pass-score-source-release-snapshot-version-not-specified","id":"agents-last-exam-pass-score-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gU2NvcmU","name":"Qwen/Qwen3.8-2.4T-A95B","notes":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Score"},{"benchmarkId":"automation-bench-pass-1-source-release-snapshot-version-not-specified","id":"automation-bench-pass-1-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","name":"Qwen/Qwen3.8-2.4T-A95B","notes":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"benchmarkId":"toolathlon-verified-pass-1-source-release-snapshot-version-not-specified","id":"toolathlon-verified-pass-1-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","name":"Qwen/Qwen3.8-2.4T-A95B","notes":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"benchmarkId":"widesearch-source-release-snapshot-version-not-specified","id":"widesearch-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gSXRlbS1GMSBvdmVyNHJ1bnM7IFF3ZW4tQWdlbnQgZm9yIFF3ZW4sQ2xhdWRlQ29kZSBmb3IgY29tcGFyYXRvcnMu","name":"Qwen/Qwen3.8-2.4T-A95B","notes":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Item-F1 over4runs; Qwen-Agent for Qwen,ClaudeCode for comparators."},{"benchmarkId":"hle-w-tools-source-release-snapshot-version-not-specified-pct","id":"hle-w-tools-source-release-snapshot-version-not-specified-pct:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","name":"Qwen/Qwen3.8-2.4T-A95B","notes":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"benchmarkId":"gpqa-diamond-source-release-snapshot-version-not-specified","id":"gpqa-diamond-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","name":"Qwen/Qwen3.8-2.4T-A95B","notes":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"benchmarkId":"hle-source-release-snapshot-version-not-specified","id":"hle-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","name":"Qwen/Qwen3.8-2.4T-A95B","notes":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"benchmarkId":"ifbench-source-release-snapshot-version-not-specified","id":"ifbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","name":"Qwen/Qwen3.8-2.4T-A95B","notes":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"benchmarkId":"onemillion-bench-expert-score-source-release-snapshot-version-not-specified","id":"onemillion-bench-expert-score-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","name":"Qwen/Qwen3.8-2.4T-A95B","notes":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"benchmarkId":"healthbench-source-release-snapshot-version-not-specified","id":"healthbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","name":"Qwen/Qwen3.8-2.4T-A95B","notes":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"benchmarkId":"plawbench-source-release-snapshot-version-not-specified","id":"plawbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","name":"Qwen/Qwen3.8-2.4T-A95B","notes":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"benchmarkId":"prbench-legal-source-release-snapshot-version-not-specified","id":"prbench-legal-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","name":"Qwen/Qwen3.8-2.4T-A95B","notes":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"benchmarkId":"prbench-finance-source-release-snapshot-version-not-specified","id":"prbench-finance-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","name":"Qwen/Qwen3.8-2.4T-A95B","notes":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"benchmarkId":"mrcr-v2-256k-8-needle-source-release-snapshot-version-not-specified","id":"mrcr-v2-256k-8-needle-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","name":"Qwen/Qwen3.8-2.4T-A95B","notes":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"benchmarkId":"longbench-v2-source-release-snapshot-version-not-specified","id":"longbench-v2-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","name":"Qwen/Qwen3.8-2.4T-A95B","notes":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"benchmarkId":"gpqa-diamond-source-release-snapshot-version-not-specified","id":"gpqa-diamond-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","name":"moonshotai/Kimi-K3","notes":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"benchmarkId":"critpt-source-release-snapshot-version-not-specified","id":"critpt-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","name":"moonshotai/Kimi-K3","notes":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"benchmarkId":"aa-lcr-source-release-snapshot-version-not-specified","id":"aa-lcr-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","name":"moonshotai/Kimi-K3","notes":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"benchmarkId":"hle-full-source-release-snapshot-version-not-specified","id":"hle-full-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aG91dCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","name":"moonshotai/Kimi-K3","notes":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"benchmarkId":"hle-full-source-release-snapshot-version-not-specified","id":"hle-full-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","name":"moonshotai/Kimi-K3","notes":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"benchmarkId":"deepswe-source-release-snapshot-version-not-specified","id":"deepswe-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","name":"moonshotai/Kimi-K3","notes":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"benchmarkId":"programbench-source-release-snapshot-version-not-specified","id":"programbench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","name":"moonshotai/Kimi-K3","notes":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"benchmarkId":"terminal-bench-2-1-pct","id":"terminal-bench-2-1-pct:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","name":"moonshotai/Kimi-K3","notes":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"benchmarkId":"frontierswe-source-release-snapshot-version-not-specified","id":"frontierswe-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","name":"moonshotai/Kimi-K3","notes":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"benchmarkId":"swe-marathon-source-release-snapshot-version-not-specified","id":"swe-marathon-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","name":"moonshotai/Kimi-K3","notes":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"benchmarkId":"posttrainbench-source-release-snapshot-version-not-specified","id":"posttrainbench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","name":"moonshotai/Kimi-K3","notes":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"benchmarkId":"mls-bench-lite-source-release-snapshot-version-not-specified","id":"mls-bench-lite-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","name":"moonshotai/Kimi-K3","notes":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"benchmarkId":"scicode-source-release-snapshot-version-not-specified","id":"scicode-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","name":"moonshotai/Kimi-K3","notes":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"benchmarkId":"kimi-code-bench-2-0","id":"kimi-code-bench-2-0:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","name":"moonshotai/Kimi-K3","notes":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"benchmarkId":"browsecomp-source-release-snapshot-version-not-specified","id":"browsecomp-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","name":"moonshotai/Kimi-K3","notes":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"benchmarkId":"deepsearchqa-f1-source-release-snapshot-version-not-specified","id":"deepsearchqa-f1-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","name":"moonshotai/Kimi-K3","notes":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"benchmarkId":"researchrubrics-source-release-snapshot-version-not-specified","id":"researchrubrics-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","name":"moonshotai/Kimi-K3","notes":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"benchmarkId":"gdpval-aa-v2-elo-source-release-snapshot-version-not-specified","id":"gdpval-aa-v2-elo-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","name":"moonshotai/Kimi-K3","notes":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"benchmarkId":"toolathlon-verified-source-release-snapshot-version-not-specified-pct","id":"toolathlon-verified-source-release-snapshot-version-not-specified-pct:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","name":"moonshotai/Kimi-K3","notes":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"benchmarkId":"mcpmark-verified-source-release-snapshot-version-not-specified","id":"mcpmark-verified-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","name":"moonshotai/Kimi-K3","notes":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"benchmarkId":"mcp-atlas-source-release-snapshot-version-not-specified","id":"mcp-atlas-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","name":"moonshotai/Kimi-K3","notes":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"benchmarkId":"automationbench-source-release-snapshot-version-not-specified","id":"automationbench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","name":"moonshotai/Kimi-K3","notes":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"benchmarkId":"jobbench-source-release-snapshot-version-not-specified","id":"jobbench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","name":"moonshotai/Kimi-K3","notes":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"benchmarkId":"aa-briefcase-elo-source-release-snapshot-version-not-specified","id":"aa-briefcase-elo-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","name":"moonshotai/Kimi-K3","notes":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"benchmarkId":"agents-last-exam-source-release-snapshot-version-not-specified","id":"agents-last-exam-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","name":"moonshotai/Kimi-K3","notes":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"benchmarkId":"apex-agents-source-release-snapshot-version-not-specified","id":"apex-agents-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","name":"moonshotai/Kimi-K3","notes":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"benchmarkId":"officeqa-pro-source-release-snapshot-version-not-specified","id":"officeqa-pro-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","name":"moonshotai/Kimi-K3","notes":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"benchmarkId":"spreadsheetbench-2-source-release-snapshot-version-not-specified","id":"spreadsheetbench-2-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","name":"moonshotai/Kimi-K3","notes":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"benchmarkId":"osworld-verified-source-release-snapshot-version-not-specified-pct","id":"osworld-verified-source-release-snapshot-version-not-specified-pct:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","name":"moonshotai/Kimi-K3","notes":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"benchmarkId":"osworld-2-0","id":"osworld-2-0:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","name":"moonshotai/Kimi-K3","notes":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"benchmarkId":"saas-bench-source-release-snapshot-version-not-specified","id":"saas-bench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","name":"moonshotai/Kimi-K3","notes":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"benchmarkId":"tau3-banking-source-release-snapshot-version-not-specified-pct","id":"tau3-banking-source-release-snapshot-version-not-specified-pct:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","name":"moonshotai/Kimi-K3","notes":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"benchmarkId":"harvey-lab-aa-source-release-snapshot-version-not-specified","id":"harvey-lab-aa-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","name":"moonshotai/Kimi-K3","notes":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"benchmarkId":"corpfin-v2-source-release-snapshot-version-not-specified","id":"corpfin-v2-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","name":"moonshotai/Kimi-K3","notes":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"benchmarkId":"finance-agent-v2-source-release-snapshot-version-not-specified","id":"finance-agent-v2-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","name":"moonshotai/Kimi-K3","notes":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"benchmarkId":"legal-research-bench-source-release-snapshot-version-not-specified","id":"legal-research-bench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","name":"moonshotai/Kimi-K3","notes":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"benchmarkId":"worldvqa-forceanswer-source-release-snapshot-version-not-specified","id":"worldvqa-forceanswer-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","name":"moonshotai/Kimi-K3","notes":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"benchmarkId":"omnidocbench-source-release-snapshot-version-not-specified","id":"omnidocbench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","name":"moonshotai/Kimi-K3","notes":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"benchmarkId":"perceptionbench-source-release-snapshot-version-not-specified","id":"perceptionbench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","name":"moonshotai/Kimi-K3","notes":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"benchmarkId":"video-mme-w-sub-source-release-snapshot-version-not-specified","id":"video-mme-w-sub-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","name":"moonshotai/Kimi-K3","notes":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"benchmarkId":"mmvu-source-release-snapshot-version-not-specified","id":"mmvu-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","name":"moonshotai/Kimi-K3","notes":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"benchmarkId":"babyvision-w-python-source-release-snapshot-version-not-specified","id":"babyvision-w-python-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","name":"moonshotai/Kimi-K3","notes":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"benchmarkId":"mmmu-pro-source-release-snapshot-version-not-specified","id":"mmmu-pro-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aG91dCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","name":"moonshotai/Kimi-K3","notes":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"benchmarkId":"mmmu-pro-source-release-snapshot-version-not-specified","id":"mmmu-pro-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","name":"moonshotai/Kimi-K3","notes":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"benchmarkId":"charxiv-rq-source-release-snapshot-version-not-specified","id":"charxiv-rq-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aG91dCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","name":"moonshotai/Kimi-K3","notes":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"benchmarkId":"charxiv-rq-source-release-snapshot-version-not-specified","id":"charxiv-rq-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","name":"moonshotai/Kimi-K3","notes":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"benchmarkId":"mathvision-source-release-snapshot-version-not-specified","id":"mathvision-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aG91dCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","name":"moonshotai/Kimi-K3","notes":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"benchmarkId":"mathvision-source-release-snapshot-version-not-specified","id":"mathvision-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","name":"moonshotai/Kimi-K3","notes":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"benchmarkId":"zerobench-pass-5-source-release-snapshot-version-not-specified","id":"zerobench-pass-5-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aG91dCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","name":"moonshotai/Kimi-K3","notes":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"benchmarkId":"zerobench-pass-5-source-release-snapshot-version-not-specified","id":"zerobench-pass-5-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","name":"moonshotai/Kimi-K3","notes":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"benchmarkId":"terminal-bench-2-1","id":"terminal-bench-2-1:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","name":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","notes":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"benchmarkId":"gdpval-source-release-snapshot-version-not-specified","id":"gdpval-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","name":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","notes":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"benchmarkId":"swe-bench-verified-source-release-snapshot-version-not-specified","id":"swe-bench-verified-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","name":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","notes":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"benchmarkId":"swe-bench-multilingual-source-release-snapshot-version-not-specified","id":"swe-bench-multilingual-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","name":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","notes":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"benchmarkId":"profbench-search-source-release-snapshot-version-not-specified","id":"profbench-search-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","name":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","notes":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"benchmarkId":"pinchbench-source-release-snapshot-version-not-specified","id":"pinchbench-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","name":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","notes":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"benchmarkId":"taubench-v3-airline-source-release-snapshot-version-not-specified","id":"taubench-v3-airline-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","name":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","notes":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"benchmarkId":"taubench-v3-retail-source-release-snapshot-version-not-specified","id":"taubench-v3-retail-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","name":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","notes":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"benchmarkId":"taubench-v3-telecom-source-release-snapshot-version-not-specified","id":"taubench-v3-telecom-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","name":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","notes":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"benchmarkId":"taubench-v3-banking-source-release-snapshot-version-not-specified","id":"taubench-v3-banking-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","name":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","notes":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"benchmarkId":"taubench-v3-average-source-release-snapshot-version-not-specified","id":"taubench-v3-average-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","name":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","notes":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"benchmarkId":"browsecomp-source-release-snapshot-version-not-specified","id":"browsecomp-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","name":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","notes":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"benchmarkId":"vals-ai-financial-agent-1-1-without-web-search-1-1","id":"vals-ai-financial-agent-1-1-without-web-search-1-1:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","name":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","notes":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"benchmarkId":"vals-ai-financial-agent-1-1-with-web-search-1-1","id":"vals-ai-financial-agent-1-1-with-web-search-1-1:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","name":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","notes":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"benchmarkId":"ioi-2025-source-release-snapshot-version-not-specified","id":"ioi-2025-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","name":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","notes":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"benchmarkId":"livecodebench-v6-source-release-snapshot-version-not-specified","id":"livecodebench-v6-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","name":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","notes":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"benchmarkId":"imoanswerbench-no-tools-source-release-snapshot-version-not-specified","id":"imoanswerbench-no-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","name":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","notes":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"benchmarkId":"imoanswerbench-with-tools-source-release-snapshot-version-not-specified","id":"imoanswerbench-with-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","name":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","notes":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"benchmarkId":"apex-shortlist-no-tools-source-release-snapshot-version-not-specified","id":"apex-shortlist-no-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","name":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","notes":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"benchmarkId":"apex-shortlist-with-tools-source-release-snapshot-version-not-specified","id":"apex-shortlist-with-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","name":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","notes":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"benchmarkId":"gpqa-no-tools-source-release-snapshot-version-not-specified","id":"gpqa-no-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","name":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","notes":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"benchmarkId":"scicode-subtask-source-release-snapshot-version-not-specified","id":"scicode-subtask-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","name":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","notes":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"benchmarkId":"hle-no-tools-source-release-snapshot-version-not-specified","id":"hle-no-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","name":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","notes":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"benchmarkId":"hle-with-tools-source-release-snapshot-version-not-specified","id":"hle-with-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","name":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","notes":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"benchmarkId":"critpt-no-tools-source-release-snapshot-version-not-specified","id":"critpt-no-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","name":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","notes":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"benchmarkId":"mmlu-pro-source-release-snapshot-version-not-specified-pct","id":"mmlu-pro-source-release-snapshot-version-not-specified-pct:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","name":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","notes":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"benchmarkId":"omniscience-accuracy-source-release-snapshot-version-not-specified","id":"omniscience-accuracy-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","name":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","notes":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"benchmarkId":"ifbench-prompt-loose-source-release-snapshot-version-not-specified","id":"ifbench-prompt-loose-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","name":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","notes":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"benchmarkId":"multi-challenge-source-release-snapshot-version-not-specified","id":"multi-challenge-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","name":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","notes":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"benchmarkId":"aa-lcr-source-release-snapshot-version-not-specified","id":"aa-lcr-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","name":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","notes":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"benchmarkId":"ruler-1m-source-release-snapshot-version-not-specified","id":"ruler-1m-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","name":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","notes":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"benchmarkId":"longbench-v2-lte-1m-source-release-snapshot-version-not-specified","id":"longbench-v2-lte-1m-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","name":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","notes":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"benchmarkId":"mmlu-prox-avg-en-de-fr-es-it-ja-zh-hi-pt-ko-source-release-snapshot-version-not-specified","id":"mmlu-prox-avg-en-de-fr-es-it-ja-zh-hi-pt-ko-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","name":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","notes":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"benchmarkId":"wmt24-en-to-xx-source-release-snapshot-version-not-specified","id":"wmt24-en-to-xx-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","name":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","notes":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"benchmarkId":"hle-wo-w-tools-source-release-snapshot-version-not-specified","id":"hle-wo-w-tools-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQuIHdpdGhvdXQgdG9vbHM","name":"deepseek-ai/DeepSeek-V4-Pro-0813","notes":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card. without tools"},{"benchmarkId":"hle-wo-w-tools-source-release-snapshot-version-not-specified","id":"hle-wo-w-tools-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQuIHdpdGggdG9vbHM","name":"deepseek-ai/DeepSeek-V4-Pro-0813","notes":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card. with tools"},{"benchmarkId":"terminal-bench-2-1","id":"terminal-bench-2-1:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","name":"deepseek-ai/DeepSeek-V4-Pro-0813","notes":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"benchmarkId":"nl2repo-source-release-snapshot-version-not-specified","id":"nl2repo-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","name":"deepseek-ai/DeepSeek-V4-Pro-0813","notes":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"benchmarkId":"cybergym-source-release-snapshot-version-not-specified-pct","id":"cybergym-source-release-snapshot-version-not-specified-pct:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","name":"deepseek-ai/DeepSeek-V4-Pro-0813","notes":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"benchmarkId":"deepswe-source-release-snapshot-version-not-specified","id":"deepswe-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","name":"deepseek-ai/DeepSeek-V4-Pro-0813","notes":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"benchmarkId":"toolathlon-verified-source-release-snapshot-version-not-specified-pct","id":"toolathlon-verified-source-release-snapshot-version-not-specified-pct:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","name":"deepseek-ai/DeepSeek-V4-Pro-0813","notes":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"benchmarkId":"agents-last-exam-source-release-snapshot-version-not-specified","id":"agents-last-exam-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","name":"deepseek-ai/DeepSeek-V4-Pro-0813","notes":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"benchmarkId":"automationbench-public-source-release-snapshot-version-not-specified","id":"automationbench-public-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","name":"deepseek-ai/DeepSeek-V4-Pro-0813","notes":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"benchmarkId":"dsbench-fullstack-source-release-snapshot-version-not-specified","id":"dsbench-fullstack-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","name":"deepseek-ai/DeepSeek-V4-Pro-0813","notes":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"benchmarkId":"dsbench-hard-source-release-snapshot-version-not-specified","id":"dsbench-hard-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","name":"deepseek-ai/DeepSeek-V4-Pro-0813","notes":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"benchmarkId":"terminal-bench-2-1","id":"terminal-bench-2-1:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","name":"zai-org/GLM-5.3","notes":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"benchmarkId":"terminal-bench-3-0","id":"terminal-bench-3-0:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","name":"zai-org/GLM-5.3","notes":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"benchmarkId":"deepswe-1-1","id":"deepswe-1-1:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","name":"zai-org/GLM-5.3","notes":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"benchmarkId":"nl2repo-source-release-snapshot-version-not-specified","id":"nl2repo-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","name":"zai-org/GLM-5.3","notes":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"benchmarkId":"programbench-almost-solved-source-release-snapshot-version-not-specified","id":"programbench-almost-solved-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","name":"zai-org/GLM-5.3","notes":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"benchmarkId":"frontierswe-source-release-snapshot-version-not-specified","id":"frontierswe-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","name":"zai-org/GLM-5.3","notes":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"benchmarkId":"swe-marathon-1-1","id":"swe-marathon-1-1:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","name":"zai-org/GLM-5.3","notes":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"benchmarkId":"posttrainbench-source-release-snapshot-version-not-specified","id":"posttrainbench-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","name":"zai-org/GLM-5.3","notes":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"benchmarkId":"cybergym-source-release-snapshot-version-not-specified","id":"cybergym-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","name":"zai-org/GLM-5.3","notes":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"benchmarkId":"exploitgym-2h-6h-source-release-snapshot-version-not-specified","id":"exploitgym-2h-6h-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4gMmggU2luZ2xlLXJ1bjg2OXRhc2tzOyBBUEkgdGltZSBub3JtYWxpemVkIGJ5IG1vZGVsIFRQUyBwbHVzIG92ZXJoZWFkOyByZXBvcnRlZCBudW1iZXIgc29sdmVkOyBkb21haW4gd2hpdGVsaXN0Lg","name":"zai-org/GLM-5.3","notes":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details. 2h Single-run869tasks; API time normalized by model TPS plus overhead; reported number solved; domain whitelist."},{"benchmarkId":"exploitgym-2h-6h-source-release-snapshot-version-not-specified","id":"exploitgym-2h-6h-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4gNmggU2luZ2xlLXJ1bjg2OXRhc2tzOyBBUEkgdGltZSBub3JtYWxpemVkIGJ5IG1vZGVsIFRQUyBwbHVzIG92ZXJoZWFkOyByZXBvcnRlZCBudW1iZXIgc29sdmVkOyBkb21haW4gd2hpdGVsaXN0Lg","name":"zai-org/GLM-5.3","notes":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details. 6h Single-run869tasks; API time normalized by model TPS plus overhead; reported number solved; domain whitelist."},{"benchmarkId":"exploitbench-source-release-snapshot-version-not-specified","id":"exploitbench-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","name":"zai-org/GLM-5.3","notes":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"benchmarkId":"toolathlon-verified-source-release-snapshot-version-not-specified","id":"toolathlon-verified-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","name":"zai-org/GLM-5.3","notes":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"benchmarkId":"automationbench-1-0-6","id":"automationbench-1-0-6:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","name":"zai-org/GLM-5.3","notes":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"benchmarkId":"agents-last-exam-ale-cli-source-release-snapshot-version-not-specified","id":"agents-last-exam-ale-cli-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","name":"zai-org/GLM-5.3","notes":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"benchmarkId":"hle-w-tools-source-release-snapshot-version-not-specified","id":"hle-w-tools-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","name":"zai-org/GLM-5.3","notes":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"benchmarkId":"gdpval-aa-v2-source-release-snapshot-version-not-specified","id":"gdpval-aa-v2-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","name":"zai-org/GLM-5.3","notes":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"benchmarkId":"aa-intelligence-index-source-release-snapshot-version-not-specified","id":"aa-intelligence-index-source-release-snapshot-version-not-specified:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","name":"Grok 4.6 launch evaluations","notes":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI."},{"benchmarkId":"gdpval-aa-v2-elo-source-release-snapshot-version-not-specified","id":"gdpval-aa-v2-elo-source-release-snapshot-version-not-specified:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","name":"Grok 4.6 launch evaluations","notes":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI."},{"benchmarkId":"cursorbench-v3-2-3-2","id":"cursorbench-v3-2-3-2:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","name":"Grok 4.6 launch evaluations","notes":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI."},{"benchmarkId":"deepswe-v1-1-1-1-pct","id":"deepswe-v1-1-1-1-pct:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","name":"Grok 4.6 launch evaluations","notes":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI."},{"benchmarkId":"frontiercode-v1-1-extended-1-1","id":"frontiercode-v1-1-extended-1-1:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","name":"Grok 4.6 launch evaluations","notes":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI."},{"benchmarkId":"apex-agents-source-release-snapshot-version-not-specified","id":"apex-agents-source-release-snapshot-version-not-specified:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","name":"Grok 4.6 launch evaluations","notes":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI."},{"benchmarkId":"terminal-bench-v3-0-3-0","id":"terminal-bench-v3-0-3-0:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","name":"Grok 4.6 launch evaluations","notes":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI."},{"benchmarkId":"apex-swe-source-release-snapshot-version-not-specified","id":"apex-swe-source-release-snapshot-version-not-specified:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","name":"Grok 4.6 launch evaluations","notes":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI."},{"benchmarkId":"aa-briefcase-elo-source-release-snapshot-version-not-specified","id":"aa-briefcase-elo-source-release-snapshot-version-not-specified:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","name":"Grok 4.6 launch evaluations","notes":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI."},{"benchmarkId":"harvey-lab-vals-source-release-snapshot-version-not-specified","id":"harvey-lab-vals-source-release-snapshot-version-not-specified:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","name":"Grok 4.6 launch evaluations","notes":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI."},{"benchmarkId":"finance-agent-v2-source-release-snapshot-version-not-specified","id":"finance-agent-v2-source-release-snapshot-version-not-specified:google:R29vZ2xlIGxhdW5jaCBjaGFydDsgYmVuY2htYXJrIG1ldGhvZG9sb2d5IGxpbmtlZCBvbiBwYWdlOyByZXBvcnRlZCBjb21wYXJhdG9yIHNldHRpbmdzIHZhcnku","name":"Gemini 3.8 Flash launch performance","notes":"Google launch chart; benchmark methodology linked on page; reported comparator settings vary."},{"benchmarkId":"harvey-legal-agent-benchmark-source-release-snapshot-version-not-specified","id":"harvey-legal-agent-benchmark-source-release-snapshot-version-not-specified:google:R29vZ2xlIGxhdW5jaCBjaGFydDsgYmVuY2htYXJrIG1ldGhvZG9sb2d5IGxpbmtlZCBvbiBwYWdlOyByZXBvcnRlZCBjb21wYXJhdG9yIHNldHRpbmdzIHZhcnku","name":"Gemini 3.8 Flash launch performance","notes":"Google launch chart; benchmark methodology linked on page; reported comparator settings vary."},{"benchmarkId":"hle-verified-source-release-snapshot-version-not-specified","id":"hle-verified-source-release-snapshot-version-not-specified:google:R29vZ2xlIGxhdW5jaCBjaGFydDsgYmVuY2htYXJrIG1ldGhvZG9sb2d5IGxpbmtlZCBvbiBwYWdlOyByZXBvcnRlZCBjb21wYXJhdG9yIHNldHRpbmdzIHZhcnku","name":"Gemini 3.8 Flash launch performance","notes":"Google launch chart; benchmark methodology linked on page; reported comparator settings vary."},{"benchmarkId":"aime-2025-source-release-snapshot-version-not-specified","id":"aime-2025-source-release-snapshot-version-not-specified:microsoft:TUFJIGF2ZyA0IHJ1bnMsIHRlbXBlcmF0dXJlIDEsIHRvcC1wIC45Nzsgc2ltcGxlIFJlQWN0IGJhc2gvc3RyaW5nLXJlcGxhY2U7IFRlcm1pbmFsLUJlbmNoIGlnbm9yZXMgcHJlZGVmaW5lZCB0aW1lb3V0cy4gQ29tcGFyYXRvciBjb25maWd1cmF0aW9ucyBmcm9tIGNpdGVkIHJlcG9ydHMu","name":"MAI-Thinking-1 technical report","notes":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports."},{"benchmarkId":"aime-2026-source-release-snapshot-version-not-specified","id":"aime-2026-source-release-snapshot-version-not-specified:microsoft:TUFJIGF2ZyA0IHJ1bnMsIHRlbXBlcmF0dXJlIDEsIHRvcC1wIC45Nzsgc2ltcGxlIFJlQWN0IGJhc2gvc3RyaW5nLXJlcGxhY2U7IFRlcm1pbmFsLUJlbmNoIGlnbm9yZXMgcHJlZGVmaW5lZCB0aW1lb3V0cy4gQ29tcGFyYXRvciBjb25maWd1cmF0aW9ucyBmcm9tIGNpdGVkIHJlcG9ydHMu","name":"MAI-Thinking-1 technical report","notes":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports."},{"benchmarkId":"hmmt-february-2026-source-release-snapshot-version-not-specified","id":"hmmt-february-2026-source-release-snapshot-version-not-specified:microsoft:TUFJIGF2ZyA0IHJ1bnMsIHRlbXBlcmF0dXJlIDEsIHRvcC1wIC45Nzsgc2ltcGxlIFJlQWN0IGJhc2gvc3RyaW5nLXJlcGxhY2U7IFRlcm1pbmFsLUJlbmNoIGlnbm9yZXMgcHJlZGVmaW5lZCB0aW1lb3V0cy4gQ29tcGFyYXRvciBjb25maWd1cmF0aW9ucyBmcm9tIGNpdGVkIHJlcG9ydHMu","name":"MAI-Thinking-1 technical report","notes":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports."},{"benchmarkId":"gpqa-diamond-source-release-snapshot-version-not-specified","id":"gpqa-diamond-source-release-snapshot-version-not-specified:microsoft:TUFJIGF2ZyA0IHJ1bnMsIHRlbXBlcmF0dXJlIDEsIHRvcC1wIC45Nzsgc2ltcGxlIFJlQWN0IGJhc2gvc3RyaW5nLXJlcGxhY2U7IFRlcm1pbmFsLUJlbmNoIGlnbm9yZXMgcHJlZGVmaW5lZCB0aW1lb3V0cy4gQ29tcGFyYXRvciBjb25maWd1cmF0aW9ucyBmcm9tIGNpdGVkIHJlcG9ydHMu","name":"MAI-Thinking-1 technical report","notes":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports."},{"benchmarkId":"livecodebench-v6-source-release-snapshot-version-not-specified-pct","id":"livecodebench-v6-source-release-snapshot-version-not-specified-pct:microsoft:TUFJIGF2ZyA0IHJ1bnMsIHRlbXBlcmF0dXJlIDEsIHRvcC1wIC45Nzsgc2ltcGxlIFJlQWN0IGJhc2gvc3RyaW5nLXJlcGxhY2U7IFRlcm1pbmFsLUJlbmNoIGlnbm9yZXMgcHJlZGVmaW5lZCB0aW1lb3V0cy4gQ29tcGFyYXRvciBjb25maWd1cmF0aW9ucyBmcm9tIGNpdGVkIHJlcG9ydHMu","name":"MAI-Thinking-1 technical report","notes":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports."},{"benchmarkId":"terminal-bench-2-0","id":"terminal-bench-2-0:microsoft:TUFJIGF2ZyA0IHJ1bnMsIHRlbXBlcmF0dXJlIDEsIHRvcC1wIC45Nzsgc2ltcGxlIFJlQWN0IGJhc2gvc3RyaW5nLXJlcGxhY2U7IFRlcm1pbmFsLUJlbmNoIGlnbm9yZXMgcHJlZGVmaW5lZCB0aW1lb3V0cy4gQ29tcGFyYXRvciBjb25maWd1cmF0aW9ucyBmcm9tIGNpdGVkIHJlcG9ydHMu","name":"MAI-Thinking-1 technical report","notes":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports."},{"benchmarkId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct","id":"swe-bench-verified-source-release-snapshot-version-not-specified-pct:microsoft:TUFJIGF2ZyA0IHJ1bnMsIHRlbXBlcmF0dXJlIDEsIHRvcC1wIC45Nzsgc2ltcGxlIFJlQWN0IGJhc2gvc3RyaW5nLXJlcGxhY2U7IFRlcm1pbmFsLUJlbmNoIGlnbm9yZXMgcHJlZGVmaW5lZCB0aW1lb3V0cy4gQ29tcGFyYXRvciBjb25maWd1cmF0aW9ucyBmcm9tIGNpdGVkIHJlcG9ydHMu","name":"MAI-Thinking-1 technical report","notes":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports."},{"benchmarkId":"swe-bench-pro-source-release-snapshot-version-not-specified","id":"swe-bench-pro-source-release-snapshot-version-not-specified:microsoft:TUFJIGF2ZyA0IHJ1bnMsIHRlbXBlcmF0dXJlIDEsIHRvcC1wIC45Nzsgc2ltcGxlIFJlQWN0IGJhc2gvc3RyaW5nLXJlcGxhY2U7IFRlcm1pbmFsLUJlbmNoIGlnbm9yZXMgcHJlZGVmaW5lZCB0aW1lb3V0cy4gQ29tcGFyYXRvciBjb25maWd1cmF0aW9ucyBmcm9tIGNpdGVkIHJlcG9ydHMu","name":"MAI-Thinking-1 technical report","notes":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports."},{"benchmarkId":"mmlu-pro-source-release-snapshot-version-not-specified","id":"mmlu-pro-source-release-snapshot-version-not-specified:microsoft:TWljcm9zb2Z0IG93biBldmFsdWF0aW9uIHN1aXRlOyBTb25uZXQgbWF4IHJlYXNvbmluZzsgVGFibGVzIDEyLzE5LiBMb25nQmVuY2hWMiBjYXBwZWQgMjU2ayw0MDggcXVlc3Rpb25zOyBDb3JwdXNRQSBHUFQtNS40IGhpZ2gganVkZ2U7IDQgcnVucy4","name":"MAI-Thinking-1 technical report","notes":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs."},{"benchmarkId":"simpleqa-verified-source-release-snapshot-version-not-specified","id":"simpleqa-verified-source-release-snapshot-version-not-specified:microsoft:TWljcm9zb2Z0IG93biBldmFsdWF0aW9uIHN1aXRlOyBTb25uZXQgbWF4IHJlYXNvbmluZzsgVGFibGVzIDEyLzE5LiBMb25nQmVuY2hWMiBjYXBwZWQgMjU2ayw0MDggcXVlc3Rpb25zOyBDb3JwdXNRQSBHUFQtNS40IGhpZ2gganVkZ2U7IDQgcnVucy4","name":"MAI-Thinking-1 technical report","notes":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs."},{"benchmarkId":"if-bench-source-release-snapshot-version-not-specified","id":"if-bench-source-release-snapshot-version-not-specified:microsoft:TWljcm9zb2Z0IG93biBldmFsdWF0aW9uIHN1aXRlOyBTb25uZXQgbWF4IHJlYXNvbmluZzsgVGFibGVzIDEyLzE5LiBMb25nQmVuY2hWMiBjYXBwZWQgMjU2ayw0MDggcXVlc3Rpb25zOyBDb3JwdXNRQSBHUFQtNS40IGhpZ2gganVkZ2U7IDQgcnVucy4","name":"MAI-Thinking-1 technical report","notes":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs."},{"benchmarkId":"advancedif-rubric-level-source-release-snapshot-version-not-specified","id":"advancedif-rubric-level-source-release-snapshot-version-not-specified:microsoft:TWljcm9zb2Z0IG93biBldmFsdWF0aW9uIHN1aXRlOyBTb25uZXQgbWF4IHJlYXNvbmluZzsgVGFibGVzIDEyLzE5LiBMb25nQmVuY2hWMiBjYXBwZWQgMjU2ayw0MDggcXVlc3Rpb25zOyBDb3JwdXNRQSBHUFQtNS40IGhpZ2gganVkZ2U7IDQgcnVucy4","name":"MAI-Thinking-1 technical report","notes":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs."},{"benchmarkId":"multichallenge-source-release-snapshot-version-not-specified","id":"multichallenge-source-release-snapshot-version-not-specified:microsoft:TWljcm9zb2Z0IG93biBldmFsdWF0aW9uIHN1aXRlOyBTb25uZXQgbWF4IHJlYXNvbmluZzsgVGFibGVzIDEyLzE5LiBMb25nQmVuY2hWMiBjYXBwZWQgMjU2ayw0MDggcXVlc3Rpb25zOyBDb3JwdXNRQSBHUFQtNS40IGhpZ2gganVkZ2U7IDQgcnVucy4","name":"MAI-Thinking-1 technical report","notes":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs."},{"benchmarkId":"graphwalks-128k-source-release-snapshot-version-not-specified","id":"graphwalks-128k-source-release-snapshot-version-not-specified:microsoft:TWljcm9zb2Z0IG93biBldmFsdWF0aW9uIHN1aXRlOyBTb25uZXQgbWF4IHJlYXNvbmluZzsgVGFibGVzIDEyLzE5LiBMb25nQmVuY2hWMiBjYXBwZWQgMjU2ayw0MDggcXVlc3Rpb25zOyBDb3JwdXNRQSBHUFQtNS40IGhpZ2gganVkZ2U7IDQgcnVucy4","name":"MAI-Thinking-1 technical report","notes":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs."},{"benchmarkId":"bfcl-v3-source-release-snapshot-version-not-specified","id":"bfcl-v3-source-release-snapshot-version-not-specified:microsoft:TWljcm9zb2Z0IG93biBldmFsdWF0aW9uIHN1aXRlOyBTb25uZXQgbWF4IHJlYXNvbmluZzsgVGFibGVzIDEyLzE5LiBMb25nQmVuY2hWMiBjYXBwZWQgMjU2ayw0MDggcXVlc3Rpb25zOyBDb3JwdXNRQSBHUFQtNS40IGhpZ2gganVkZ2U7IDQgcnVucy4","name":"MAI-Thinking-1 technical report","notes":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs."},{"benchmarkId":"healthbench-professional-source-release-snapshot-version-not-specified","id":"healthbench-professional-source-release-snapshot-version-not-specified:microsoft:TWljcm9zb2Z0IG93biBldmFsdWF0aW9uIHN1aXRlOyBTb25uZXQgbWF4IHJlYXNvbmluZzsgVGFibGVzIDEyLzE5LiBMb25nQmVuY2hWMiBjYXBwZWQgMjU2ayw0MDggcXVlc3Rpb25zOyBDb3JwdXNRQSBHUFQtNS40IGhpZ2gganVkZ2U7IDQgcnVucy4","name":"MAI-Thinking-1 technical report","notes":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs."},{"benchmarkId":"medxpertqa-source-release-snapshot-version-not-specified","id":"medxpertqa-source-release-snapshot-version-not-specified:microsoft:TWljcm9zb2Z0IG93biBldmFsdWF0aW9uIHN1aXRlOyBTb25uZXQgbWF4IHJlYXNvbmluZzsgVGFibGVzIDEyLzE5LiBMb25nQmVuY2hWMiBjYXBwZWQgMjU2ayw0MDggcXVlc3Rpb25zOyBDb3JwdXNRQSBHUFQtNS40IGhpZ2gganVkZ2U7IDQgcnVucy4","name":"MAI-Thinking-1 technical report","notes":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs."},{"benchmarkId":"longbenchv2-source-release-snapshot-version-not-specified","id":"longbenchv2-source-release-snapshot-version-not-specified:microsoft:TWljcm9zb2Z0IG93biBldmFsdWF0aW9uIHN1aXRlOyBTb25uZXQgbWF4IHJlYXNvbmluZzsgVGFibGVzIDEyLzE5LiBMb25nQmVuY2hWMiBjYXBwZWQgMjU2ayw0MDggcXVlc3Rpb25zOyBDb3JwdXNRQSBHUFQtNS40IGhpZ2gganVkZ2U7IDQgcnVucy4","name":"MAI-Thinking-1 technical report","notes":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs."},{"benchmarkId":"corpusqa-source-release-snapshot-version-not-specified","id":"corpusqa-source-release-snapshot-version-not-specified:microsoft:TWljcm9zb2Z0IG93biBldmFsdWF0aW9uIHN1aXRlOyBTb25uZXQgbWF4IHJlYXNvbmluZzsgVGFibGVzIDEyLzE5LiBMb25nQmVuY2hWMiBjYXBwZWQgMjU2ayw0MDggcXVlc3Rpb25zOyBDb3JwdXNRQSBHUFQtNS40IGhpZ2gganVkZ2U7IDQgcnVucy4","name":"MAI-Thinking-1 technical report","notes":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs."},{"benchmarkId":"tau2-bench-telecom-source-release-snapshot-version-not-specified","id":"tau2-bench-telecom-source-release-snapshot-version-not-specified:cohere:Q29oZXJlIGxhdW5jaCBldmFsdWF0aW9uOyBOb3J0aCBtZXRyaWMgdXNlcyBpbnRlcm5hbCBMTE0ganVkZ2Uu","name":"Command A+ launch benchmarks","notes":"Cohere launch evaluation; North metric uses internal LLM judge."},{"benchmarkId":"terminal-bench-hard-source-release-snapshot-version-not-specified","id":"terminal-bench-hard-source-release-snapshot-version-not-specified:cohere:Q29oZXJlIGxhdW5jaCBldmFsdWF0aW9uOyBOb3J0aCBtZXRyaWMgdXNlcyBpbnRlcm5hbCBMTE0ganVkZ2Uu","name":"Command A+ launch benchmarks","notes":"Cohere launch evaluation; North metric uses internal LLM judge."},{"benchmarkId":"north-memory-usage-quality-source-release-snapshot-version-not-specified","id":"north-memory-usage-quality-source-release-snapshot-version-not-specified:cohere:Q29oZXJlIGxhdW5jaCBldmFsdWF0aW9uOyBOb3J0aCBtZXRyaWMgdXNlcyBpbnRlcm5hbCBMTE0ganVkZ2Uu","name":"Command A+ launch benchmarks","notes":"Cohere launch evaluation; North metric uses internal LLM judge."},{"benchmarkId":"mmmu-pro-source-release-snapshot-version-not-specified","id":"mmmu-pro-source-release-snapshot-version-not-specified:cohere:Q29oZXJlIGxhdW5jaCBtdWx0aW1vZGFsIGV2YWx1YXRpb25zLg","name":"Command A+ launch benchmarks","notes":"Cohere launch multimodal evaluations."},{"benchmarkId":"mmmu-source-release-snapshot-version-not-specified","id":"mmmu-source-release-snapshot-version-not-specified:cohere:Q29oZXJlIGxhdW5jaCBtdWx0aW1vZGFsIGV2YWx1YXRpb25zLg","name":"Command A+ launch benchmarks","notes":"Cohere launch multimodal evaluations."},{"benchmarkId":"mathvista-source-release-snapshot-version-not-specified","id":"mathvista-source-release-snapshot-version-not-specified:cohere:Q29oZXJlIGxhdW5jaCBtdWx0aW1vZGFsIGV2YWx1YXRpb25zLg","name":"Command A+ launch benchmarks","notes":"Cohere launch multimodal evaluations."},{"benchmarkId":"charxiv-reasoning-source-release-snapshot-version-not-specified","id":"charxiv-reasoning-source-release-snapshot-version-not-specified:cohere:Q29oZXJlIGxhdW5jaCBtdWx0aW1vZGFsIGV2YWx1YXRpb25zLg","name":"Command A+ launch benchmarks","notes":"Cohere launch multimodal evaluations."},{"benchmarkId":"aime-2025-avg-16-source-release-snapshot-version-not-specified","id":"aime-2025-avg-16-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IFNvbm5ldDQuNiBleHRlcm5hbCBBUEkgdHJ1bmNhdGlvbiBjYXZlYXQu","name":"Mistral Medium 3.5 model card performance charts","notes":"Maximum reasoning; Sonnet4.6 external API truncation caveat."},{"benchmarkId":"allenai-ifbench-source-release-snapshot-version-not-specified","id":"allenai-ifbench-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IFNvbm5ldDQuNiBleHRlcm5hbCBBUEkgdHJ1bmNhdGlvbiBjYXZlYXQu","name":"Mistral Medium 3.5 model card performance charts","notes":"Maximum reasoning; Sonnet4.6 external API truncation caveat."},{"benchmarkId":"collie-source-release-snapshot-version-not-specified","id":"collie-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IFNvbm5ldDQuNiBleHRlcm5hbCBBUEkgdHJ1bmNhdGlvbiBjYXZlYXQu","name":"Mistral Medium 3.5 model card performance charts","notes":"Maximum reasoning; Sonnet4.6 external API truncation caveat."},{"benchmarkId":"beyondaime-avg-16-source-release-snapshot-version-not-specified","id":"beyondaime-avg-16-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IFNvbm5ldDQuNiBleHRlcm5hbCBBUEkgdHJ1bmNhdGlvbiBjYXZlYXQu","name":"Mistral Medium 3.5 model card performance charts","notes":"Maximum reasoning; Sonnet4.6 external API truncation caveat."},{"benchmarkId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct","id":"swe-bench-verified-source-release-snapshot-version-not-specified-pct:mistral:TWF4aW11bSByZWFzb25pbmc7IHRhdTMgNCB0cmlhbHMsIEdQVDUuMiBsb3cgc2ltdWxhdG9yIGV4Y2VwdCBTaWVycmEgcmVwb3J0ZWQgY29tcGFyaXNvbnM7IEJyb3dzZUNvbXAgZGlzY2FyZC1hbGwgY29udGV4dCBhdDEwMGsu","name":"Mistral Medium 3.5 model card performance charts","notes":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k."},{"benchmarkId":"tau3-telecom-source-release-snapshot-version-not-specified","id":"tau3-telecom-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IHRhdTMgNCB0cmlhbHMsIEdQVDUuMiBsb3cgc2ltdWxhdG9yIGV4Y2VwdCBTaWVycmEgcmVwb3J0ZWQgY29tcGFyaXNvbnM7IEJyb3dzZUNvbXAgZGlzY2FyZC1hbGwgY29udGV4dCBhdDEwMGsu","name":"Mistral Medium 3.5 model card performance charts","notes":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k."},{"benchmarkId":"tau3-airline-source-release-snapshot-version-not-specified","id":"tau3-airline-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IHRhdTMgNCB0cmlhbHMsIEdQVDUuMiBsb3cgc2ltdWxhdG9yIGV4Y2VwdCBTaWVycmEgcmVwb3J0ZWQgY29tcGFyaXNvbnM7IEJyb3dzZUNvbXAgZGlzY2FyZC1hbGwgY29udGV4dCBhdDEwMGsu","name":"Mistral Medium 3.5 model card performance charts","notes":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k."},{"benchmarkId":"tau3-retail-source-release-snapshot-version-not-specified","id":"tau3-retail-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IHRhdTMgNCB0cmlhbHMsIEdQVDUuMiBsb3cgc2ltdWxhdG9yIGV4Y2VwdCBTaWVycmEgcmVwb3J0ZWQgY29tcGFyaXNvbnM7IEJyb3dzZUNvbXAgZGlzY2FyZC1hbGwgY29udGV4dCBhdDEwMGsu","name":"Mistral Medium 3.5 model card performance charts","notes":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k."},{"benchmarkId":"tau3-banking-source-release-snapshot-version-not-specified","id":"tau3-banking-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IHRhdTMgNCB0cmlhbHMsIEdQVDUuMiBsb3cgc2ltdWxhdG9yIGV4Y2VwdCBTaWVycmEgcmVwb3J0ZWQgY29tcGFyaXNvbnM7IEJyb3dzZUNvbXAgZGlzY2FyZC1hbGwgY29udGV4dCBhdDEwMGsu","name":"Mistral Medium 3.5 model card performance charts","notes":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k."},{"benchmarkId":"browsecomp-source-release-snapshot-version-not-specified","id":"browsecomp-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IHRhdTMgNCB0cmlhbHMsIEdQVDUuMiBsb3cgc2ltdWxhdG9yIGV4Y2VwdCBTaWVycmEgcmVwb3J0ZWQgY29tcGFyaXNvbnM7IEJyb3dzZUNvbXAgZGlzY2FyZC1hbGwgY29udGV4dCBhdDEwMGsu","name":"Mistral Medium 3.5 model card performance charts","notes":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k."},{"benchmarkId":"mmlu-pro-source-release-snapshot-version-not-specified-pct","id":"mmlu-pro-source-release-snapshot-version-not-specified-pct:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","name":"Amazon Nova 2 technical report","notes":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"benchmarkId":"gpqa-diamond-source-release-snapshot-version-not-specified","id":"gpqa-diamond-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","name":"Amazon Nova 2 technical report","notes":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"benchmarkId":"aime-2025-source-release-snapshot-version-not-specified","id":"aime-2025-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","name":"Amazon Nova 2 technical report","notes":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"benchmarkId":"ifbench-prompt-loose-source-release-snapshot-version-not-specified-pct","id":"ifbench-prompt-loose-source-release-snapshot-version-not-specified-pct:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","name":"Amazon Nova 2 technical report","notes":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"benchmarkId":"multichallenge-source-release-snapshot-version-not-specified","id":"multichallenge-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","name":"Amazon Nova 2 technical report","notes":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"benchmarkId":"longcodebench-1m-source-release-snapshot-version-not-specified","id":"longcodebench-1m-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","name":"Amazon Nova 2 technical report","notes":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"benchmarkId":"tau2-telecom-source-release-snapshot-version-not-specified","id":"tau2-telecom-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","name":"Amazon Nova 2 technical report","notes":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"benchmarkId":"bfcl-v4-source-release-snapshot-version-not-specified","id":"bfcl-v4-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","name":"Amazon Nova 2 technical report","notes":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"benchmarkId":"tau2-retail-verified-source-release-snapshot-version-not-specified","id":"tau2-retail-verified-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","name":"Amazon Nova 2 technical report","notes":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"benchmarkId":"tau2-airline-verified-source-release-snapshot-version-not-specified","id":"tau2-airline-verified-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","name":"Amazon Nova 2 technical report","notes":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"benchmarkId":"mcpatlas-source-release-snapshot-version-not-specified","id":"mcpatlas-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","name":"Amazon Nova 2 technical report","notes":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"benchmarkId":"mmmu-pro-source-release-snapshot-version-not-specified","id":"mmmu-pro-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","name":"Amazon Nova 2 technical report","notes":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"benchmarkId":"ocrbench-v2-average-accuracy-source-release-snapshot-version-not-specified","id":"ocrbench-v2-average-accuracy-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","name":"Amazon Nova 2 technical report","notes":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"benchmarkId":"realkie-fcc-verified-alns-source-release-snapshot-version-not-specified","id":"realkie-fcc-verified-alns-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","name":"Amazon Nova 2 technical report","notes":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"benchmarkId":"qvhighlights-r1-0-5","id":"qvhighlights-r1-0-5:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","name":"Amazon Nova 2 technical report","notes":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"benchmarkId":"screenspot-point-accuracy-source-release-snapshot-version-not-specified","id":"screenspot-point-accuracy-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","name":"Amazon Nova 2 technical report","notes":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"benchmarkId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct","id":"swe-bench-verified-source-release-snapshot-version-not-specified-pct:amazon:Tm92YTIgaW50ZXJuYWwgYWdlbnRpYyBzY2FmZm9sZDsgc2NhbGVkIGluZmVyZW5jZSBleHBsaWNpdGx5IHNlcGFyYXRlLiBDb21wYXJhdG9yIHNldHVwcyBmcm9tIGNpdGVkIHByb3ZpZGVyIHJlcG9ydC4","name":"Amazon Nova 2 technical report","notes":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report."},{"benchmarkId":"terminal-bench-1-0","id":"terminal-bench-1-0:amazon:Tm92YTIgaW50ZXJuYWwgYWdlbnRpYyBzY2FmZm9sZDsgc2NhbGVkIGluZmVyZW5jZSBleHBsaWNpdGx5IHNlcGFyYXRlLiBDb21wYXJhdG9yIHNldHVwcyBmcm9tIGNpdGVkIHByb3ZpZGVyIHJlcG9ydC4","name":"Amazon Nova 2 technical report","notes":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report."},{"benchmarkId":"livecodebench-v5-july2024-january2025-source-release-snapshot-version-not-specified","id":"livecodebench-v5-july2024-january2025-source-release-snapshot-version-not-specified:amazon:Tm92YTIgaW50ZXJuYWwgYWdlbnRpYyBzY2FmZm9sZDsgc2NhbGVkIGluZmVyZW5jZSBleHBsaWNpdGx5IHNlcGFyYXRlLiBDb21wYXJhdG9yIHNldHVwcyBmcm9tIGNpdGVkIHByb3ZpZGVyIHJlcG9ydC4","name":"Amazon Nova 2 technical report","notes":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report."},{"benchmarkId":"swe-bench-verified-scaled-inference-source-release-snapshot-version-not-specified","id":"swe-bench-verified-scaled-inference-source-release-snapshot-version-not-specified:amazon:Tm92YTIgaW50ZXJuYWwgYWdlbnRpYyBzY2FmZm9sZDsgc2NhbGVkIGluZmVyZW5jZSBleHBsaWNpdGx5IHNlcGFyYXRlLiBDb21wYXJhdG9yIHNldHVwcyBmcm9tIGNpdGVkIHByb3ZpZGVyIHJlcG9ydC4","name":"Amazon Nova 2 technical report","notes":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report."},{"benchmarkId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct","id":"swe-bench-verified-source-release-snapshot-version-not-specified-pct:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","name":"MiniMax M3 model card benchmark figure","notes":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"benchmarkId":"swe-bench-pro-source-release-snapshot-version-not-specified","id":"swe-bench-pro-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","name":"MiniMax M3 model card benchmark figure","notes":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"benchmarkId":"terminal-bench-2-1-pct","id":"terminal-bench-2-1-pct:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","name":"MiniMax M3 model card benchmark figure","notes":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"benchmarkId":"sweatlas-qna-source-release-snapshot-version-not-specified","id":"sweatlas-qna-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","name":"MiniMax M3 model card benchmark figure","notes":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"benchmarkId":"nl2repo-source-release-snapshot-version-not-specified","id":"nl2repo-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","name":"MiniMax M3 model card benchmark figure","notes":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"benchmarkId":"sweatlas-testwriting-source-release-snapshot-version-not-specified","id":"sweatlas-testwriting-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","name":"MiniMax M3 model card benchmark figure","notes":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"benchmarkId":"swe-fficiency-source-release-snapshot-version-not-specified","id":"swe-fficiency-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","name":"MiniMax M3 model card benchmark figure","notes":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"benchmarkId":"livesqlbench-source-release-snapshot-version-not-specified","id":"livesqlbench-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","name":"MiniMax M3 model card benchmark figure","notes":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"benchmarkId":"cl-bench-source-release-snapshot-version-not-specified","id":"cl-bench-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","name":"MiniMax M3 model card benchmark figure","notes":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"benchmarkId":"vibe-v2-source-release-snapshot-version-not-specified","id":"vibe-v2-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","name":"MiniMax M3 model card benchmark figure","notes":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"benchmarkId":"svg-bench-source-release-snapshot-version-not-specified","id":"svg-bench-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","name":"MiniMax M3 model card benchmark figure","notes":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"benchmarkId":"posttrainbench-source-release-snapshot-version-not-specified","id":"posttrainbench-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","name":"MiniMax M3 model card benchmark figure","notes":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"benchmarkId":"kernelbench-hard-source-release-snapshot-version-not-specified","id":"kernelbench-hard-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","name":"MiniMax M3 model card benchmark figure","notes":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"benchmarkId":"paperbench-source-release-snapshot-version-not-specified","id":"paperbench-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","name":"MiniMax M3 model card benchmark figure","notes":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"benchmarkId":"browsecomp-source-release-snapshot-version-not-specified","id":"browsecomp-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","name":"MiniMax M3 model card benchmark figure","notes":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"benchmarkId":"draco-source-release-snapshot-version-not-specified","id":"draco-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","name":"MiniMax M3 model card benchmark figure","notes":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"benchmarkId":"gdpval-rubrics-source-release-snapshot-version-not-specified","id":"gdpval-rubrics-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","name":"MiniMax M3 model card benchmark figure","notes":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"benchmarkId":"bankertoolbench-source-release-snapshot-version-not-specified","id":"bankertoolbench-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","name":"MiniMax M3 model card benchmark figure","notes":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"benchmarkId":"officeqa-pro-source-release-snapshot-version-not-specified","id":"officeqa-pro-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","name":"MiniMax M3 model card benchmark figure","notes":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"benchmarkId":"spreadsheetbench-v1-source-release-snapshot-version-not-specified","id":"spreadsheetbench-v1-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","name":"MiniMax M3 model card benchmark figure","notes":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"benchmarkId":"loca-bench-256k-source-release-snapshot-version-not-specified","id":"loca-bench-256k-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","name":"MiniMax M3 model card benchmark figure","notes":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"benchmarkId":"mcpatlas-source-release-snapshot-version-not-specified","id":"mcpatlas-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","name":"MiniMax M3 model card benchmark figure","notes":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"benchmarkId":"apex-agents-source-release-snapshot-version-not-specified","id":"apex-agents-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","name":"MiniMax M3 model card benchmark figure","notes":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"benchmarkId":"claw-eval-source-release-snapshot-version-not-specified","id":"claw-eval-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","name":"MiniMax M3 model card benchmark figure","notes":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"benchmarkId":"osworld-verified-source-release-snapshot-version-not-specified","id":"osworld-verified-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","name":"MiniMax M3 model card benchmark figure","notes":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"benchmarkId":"omnidocbench-source-release-snapshot-version-not-specified","id":"omnidocbench-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","name":"MiniMax M3 model card benchmark figure","notes":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"benchmarkId":"mmmu-pro-source-release-snapshot-version-not-specified","id":"mmmu-pro-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","name":"MiniMax M3 model card benchmark figure","notes":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"benchmarkId":"videommmu-source-release-snapshot-version-not-specified","id":"videommmu-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","name":"MiniMax M3 model card benchmark figure","notes":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"benchmarkId":"videomme-with-subtitles-source-release-snapshot-version-not-specified","id":"videomme-with-subtitles-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","name":"MiniMax M3 model card benchmark figure","notes":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"benchmarkId":"yc-bench-source-release-snapshot-version-not-specified","id":"yc-bench-source-release-snapshot-version-not-specified:minimax:TGF1bmNoIGZpZ3VyZTsgYWdlbnQgZmluYWwgYXNzZXRz","name":"MiniMax M3 model card benchmark figure","notes":"Launch figure; agent final assets"},{"benchmarkId":"imo-2025-source-release-snapshot-version-not-specified-points-42","id":"imo-2025-source-release-snapshot-version-not-specified-points-42:minimax:TWluaU1heCBwb2ludHMgb3V0IG9mNDI7IGNvbXBhcmF0b3IgcGVyY2VudGFnZXMgYXMgcHJpbnRlZA","name":"MiniMax M3 model card benchmark figure","notes":"MiniMax points out of42; comparator percentages as printed"},{"benchmarkId":"imo-2025-source-release-snapshot-version-not-specified","id":"imo-2025-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBwb2ludHMgb3V0IG9mNDI7IGNvbXBhcmF0b3IgcGVyY2VudGFnZXMgYXMgcHJpbnRlZA","name":"MiniMax M3 model card benchmark figure","notes":"MiniMax points out of42; comparator percentages as printed"},{"benchmarkId":"usamo-2026-source-release-snapshot-version-not-specified-points-42","id":"usamo-2026-source-release-snapshot-version-not-specified-points-42:minimax:TWluaU1heCBwb2ludHMgb3V0IG9mNDI7IGNvbXBhcmF0b3IgcGVyY2VudGFnZXMgYXMgcHJpbnRlZA","name":"MiniMax M3 model card benchmark figure","notes":"MiniMax points out of42; comparator percentages as printed"},{"benchmarkId":"usamo-2026-source-release-snapshot-version-not-specified","id":"usamo-2026-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBwb2ludHMgb3V0IG9mNDI7IGNvbXBhcmF0b3IgcGVyY2VudGFnZXMgYXMgcHJpbnRlZA","name":"MiniMax M3 model card benchmark figure","notes":"MiniMax points out of42; comparator percentages as printed"},{"benchmarkId":"hle-with-tools-source-release-snapshot-version-not-specified-pct","id":"hle-with-tools-source-release-snapshot-version-not-specified-pct:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","name":"Muse Spark 1.1 evaluation report Figure44","notes":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"benchmarkId":"hle-without-tools-source-release-snapshot-version-not-specified","id":"hle-without-tools-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","name":"Muse Spark 1.1 evaluation report Figure44","notes":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"benchmarkId":"mrcr-v2-1m-8-needle-source-release-snapshot-version-not-specified","id":"mrcr-v2-1m-8-needle-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","name":"Muse Spark 1.1 evaluation report Figure44","notes":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"benchmarkId":"mcpatlas-source-release-snapshot-version-not-specified","id":"mcpatlas-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","name":"Muse Spark 1.1 evaluation report Figure44","notes":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"benchmarkId":"toolathlon-verified-source-release-snapshot-version-not-specified","id":"toolathlon-verified-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","name":"Muse Spark 1.1 evaluation report Figure44","notes":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"benchmarkId":"osworld-verified-source-release-snapshot-version-not-specified","id":"osworld-verified-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","name":"Muse Spark 1.1 evaluation report Figure44","notes":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"benchmarkId":"osworld-2-0-binary-without-exec-2-0","id":"osworld-2-0-binary-without-exec-2-0:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","name":"Muse Spark 1.1 evaluation report Figure44","notes":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"benchmarkId":"osworld-2-0-partial-without-exec-2-0","id":"osworld-2-0-partial-without-exec-2-0:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","name":"Muse Spark 1.1 evaluation report Figure44","notes":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"benchmarkId":"webarena-verified-source-release-snapshot-version-not-specified","id":"webarena-verified-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","name":"Muse Spark 1.1 evaluation report Figure44","notes":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"benchmarkId":"deepsearchqa-source-release-snapshot-version-not-specified","id":"deepsearchqa-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","name":"Muse Spark 1.1 evaluation report Figure44","notes":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"benchmarkId":"gdpval-aa-v2-elo-source-release-snapshot-version-not-specified-elo","id":"gdpval-aa-v2-elo-source-release-snapshot-version-not-specified-elo:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","name":"Muse Spark 1.1 evaluation report Figure44","notes":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"benchmarkId":"jobbench-source-release-snapshot-version-not-specified","id":"jobbench-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","name":"Muse Spark 1.1 evaluation report Figure44","notes":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"benchmarkId":"finance-agent-v2-source-release-snapshot-version-not-specified","id":"finance-agent-v2-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","name":"Muse Spark 1.1 evaluation report Figure44","notes":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"benchmarkId":"terminal-bench-2-1-pct","id":"terminal-bench-2-1-pct:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","name":"Muse Spark 1.1 evaluation report Figure44","notes":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"benchmarkId":"swe-bench-pro-source-release-snapshot-version-not-specified","id":"swe-bench-pro-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","name":"Muse Spark 1.1 evaluation report Figure44","notes":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"benchmarkId":"deepswe-v1-1-1-1-pct","id":"deepswe-v1-1-1-1-pct:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","name":"Muse Spark 1.1 evaluation report Figure44","notes":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"benchmarkId":"healthbench-professional-source-release-snapshot-version-not-specified","id":"healthbench-professional-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","name":"Muse Spark 1.1 evaluation report Figure44","notes":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"benchmarkId":"charxiv-reasoning-with-tools-source-release-snapshot-version-not-specified","id":"charxiv-reasoning-with-tools-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","name":"Muse Spark 1.1 evaluation report Figure44","notes":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"benchmarkId":"babyvision-with-tools-source-release-snapshot-version-not-specified","id":"babyvision-with-tools-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","name":"Muse Spark 1.1 evaluation report Figure44","notes":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"benchmarkId":"tau3-telecom-source-release-snapshot-version-as-labeled","id":"tau3-telecom-source-release-snapshot-version-as-labeled:mistral:TWF4aW11bSByZWFzb25pbmc7IE1pc3RyYWwgc2FtZS1sYWIgY29tcGFyaXNvbjsgdGF1MyA0dHJpYWxzIEdQVDUuMmxvdyB1c2VyIHNpbXVsYXRvcjsgQnJvd3NlQ29tcCBjb250ZXh0LWRpc2NhcmQgc2V0dXAgaW4gY2FyZC4","name":"Mistral Medium 3.5 model card performance charts","notes":"Maximum reasoning; Mistral same-lab comparison; tau3 4trials GPT5.2low user simulator; BrowseComp context-discard setup in card."},{"benchmarkId":"tau3-airline-source-release-snapshot-version-as-labeled","id":"tau3-airline-source-release-snapshot-version-as-labeled:mistral:TWF4aW11bSByZWFzb25pbmc7IE1pc3RyYWwgc2FtZS1sYWIgY29tcGFyaXNvbjsgdGF1MyA0dHJpYWxzIEdQVDUuMmxvdyB1c2VyIHNpbXVsYXRvcjsgQnJvd3NlQ29tcCBjb250ZXh0LWRpc2NhcmQgc2V0dXAgaW4gY2FyZC4","name":"Mistral Medium 3.5 model card performance charts","notes":"Maximum reasoning; Mistral same-lab comparison; tau3 4trials GPT5.2low user simulator; BrowseComp context-discard setup in card."},{"benchmarkId":"tau3-retail-source-release-snapshot-version-as-labeled","id":"tau3-retail-source-release-snapshot-version-as-labeled:mistral:TWF4aW11bSByZWFzb25pbmc7IE1pc3RyYWwgc2FtZS1sYWIgY29tcGFyaXNvbjsgdGF1MyA0dHJpYWxzIEdQVDUuMmxvdyB1c2VyIHNpbXVsYXRvcjsgQnJvd3NlQ29tcCBjb250ZXh0LWRpc2NhcmQgc2V0dXAgaW4gY2FyZC4","name":"Mistral Medium 3.5 model card performance charts","notes":"Maximum reasoning; Mistral same-lab comparison; tau3 4trials GPT5.2low user simulator; BrowseComp context-discard setup in card."},{"benchmarkId":"tau3-banking-source-release-snapshot-version-as-labeled","id":"tau3-banking-source-release-snapshot-version-as-labeled:mistral:TWF4aW11bSByZWFzb25pbmc7IE1pc3RyYWwgc2FtZS1sYWIgY29tcGFyaXNvbjsgdGF1MyA0dHJpYWxzIEdQVDUuMmxvdyB1c2VyIHNpbXVsYXRvcjsgQnJvd3NlQ29tcCBjb250ZXh0LWRpc2NhcmQgc2V0dXAgaW4gY2FyZC4","name":"Mistral Medium 3.5 model card performance charts","notes":"Maximum reasoning; Mistral same-lab comparison; tau3 4trials GPT5.2low user simulator; BrowseComp context-discard setup in card."},{"benchmarkId":"browsecomp-source-release-snapshot-version-as-labeled","id":"browsecomp-source-release-snapshot-version-as-labeled:mistral:TWF4aW11bSByZWFzb25pbmc7IE1pc3RyYWwgc2FtZS1sYWIgY29tcGFyaXNvbjsgdGF1MyA0dHJpYWxzIEdQVDUuMmxvdyB1c2VyIHNpbXVsYXRvcjsgQnJvd3NlQ29tcCBjb250ZXh0LWRpc2NhcmQgc2V0dXAgaW4gY2FyZC4","name":"Mistral Medium 3.5 model card performance charts","notes":"Maximum reasoning; Mistral same-lab comparison; tau3 4trials GPT5.2low user simulator; BrowseComp context-discard setup in card."},{"benchmarkId":"swe-bench-verified-source-release-snapshot-version-as-labeled","id":"swe-bench-verified-source-release-snapshot-version-as-labeled:mistral:TWlzdHJhbCBwcmV2aW91cyBjb2RpbmcgbW9kZWwgY29tcGFyaXNvbiwgcHVibGlzaGVkIG1vZGVsLWNhcmQgaGFybmVzcyBzZXR0aW5ncy4","name":"Mistral Medium 3.5 model card performance charts","notes":"Mistral previous coding model comparison, published model-card harness settings."},{"benchmarkId":"ifbench-source-release-snapshot-version-as-labeled","id":"ifbench-source-release-snapshot-version-as-labeled:cohere:U2luZ2xlLXR1cm4gbG9vc2UscHJvbXB0IGFjY3VyYWN5LDI5NHByb21wdHMgeDUgcmVwZWF0cy4","name":"Command A+ launch benchmarks","notes":"Single-turn loose,prompt accuracy,294prompts x5 repeats."},{"benchmarkId":"aime-2025-source-release-snapshot-version-as-labeled","id":"aime-2025-source-release-snapshot-version-as-labeled:cohere:T2ZmaWNpYWwzMHF1ZXN0aW9ucyB4MTAgcmVwZWF0czsgcGFzc0AxLg","name":"Command A+ launch benchmarks","notes":"Official30questions x10 repeats; pass@1."},{"benchmarkId":"scicode-source-release-snapshot-version-as-labeled","id":"scicode-source-release-snapshot-version-as-labeled:cohere:NjVwcm9ibGVtcy8yODhzdWJwcm9ibGVtczsgc2NpZW50aXN0LWFubm90YXRlZCBiYWNrZ3JvdW5kLg","name":"Command A+ launch benchmarks","notes":"65problems/288subproblems; scientist-annotated background."},{"benchmarkId":"north-agentic-question-answering-source-release-snapshot-version-as-labeled","id":"north-agentic-question-answering-source-release-snapshot-version-as-labeled:cohere:SW50ZXJuYWwgTm9ydGggZW50ZXJwcmlzZSBNQ1AgY2xvdWQtZmlsZSBRQSxMTE0ganVkZ2Uu","name":"Command A+ launch benchmarks","notes":"Internal North enterprise MCP cloud-file QA,LLM judge."},{"benchmarkId":"north-data-analysis-source-release-snapshot-version-as-labeled","id":"north-data-analysis-source-release-snapshot-version-as-labeled:cohere:SW50ZXJuYWwgTm9ydGggdXBsb2FkZWQgc3ByZWFkc2hlZXQgZGF0YS1zY2llbmNlIHRhc2tzLExMTSBqdWRnZS4","name":"Command A+ launch benchmarks","notes":"Internal North uploaded spreadsheet data-science tasks,LLM judge."},{"benchmarkId":"mt-aime-2025-arabic-japanese-korean-source-release-snapshot-version-as-labeled","id":"mt-aime-2025-arabic-japanese-korean-source-release-snapshot-version-as-labeled:cohere:SW50ZXJuYWwgQ29tbWFuZCBBIFRyYW5zbGF0ZSB0cmFuc2xhdGlvbnM7IEFyYWJpYyxKYXBhbmVzZSxLb3JlYW4u","name":"Command A+ launch benchmarks","notes":"Internal Command A Translate translations; Arabic,Japanese,Korean."},{"benchmarkId":"wmt24-50-varieties-source-release-snapshot-version-as-labeled","id":"wmt24-50-varieties-source-release-snapshot-version-as-labeled:cohere:eENPTUVUeGwgYXZlcmFnZTUwIHZhcmlldGllcyxpbmNsdWRpbmcgaW50ZXJuYWwgSXJpc2gvTWFsdGVzZSB0cmFuc2xhdGlvbnMgYW5kIFNlcmJpYW4gdHJhbnNsaXRlcmF0aW9uLg","name":"Command A+ launch benchmarks","notes":"xCOMETxl average50 varieties,including internal Irish/Maltese translations and Serbian transliteration."},{"benchmarkId":"charxiv-descriptive-source-release-snapshot-version-as-labeled","id":"charxiv-descriptive-source-release-snapshot-version-as-labeled:cohere:U3RhbmRhcmQgbWV0aG9kb2xvZ3k7IGludGVnZXIgcm91bmRlZCBsYWJlbHMgaW4gY2hhcnQu","name":"Command A+ launch benchmarks","notes":"Standard methodology; integer rounded labels in chart."},{"benchmarkId":"aa-intelligence-index-source-release-snapshot-version-as-labeled","id":"aa-intelligence-index-source-release-snapshot-version-as-labeled:cohere:Q29oZXJlIHF1b3RlZCBBQSBsYXVuY2ggc25hcHNob3Q7IGluZGV4IHZlcnNpb24gdW5zcGVjaWZpZWQu","name":"Command A+ launch benchmarks","notes":"Cohere quoted AA launch snapshot; index version unspecified."},{"benchmarkId":"gdpval-aa-2","id":"gdpval-aa-2:meta-muse-1-3-report:QXJ0aWZpY2lhbCBBbmFseXNpcyBTdGlycnVwIHNoZWxsL3dlYiBoYXJuZXNzOzIyMHRhc2tzOyBibGluZCBwYWlyd2lzZSBjb21wYXJpc29uczsgcmVhc29uaW5nIG1heA","name":"Muse Spark1.3 evaluation methodology","notes":"Artificial Analysis Stirrup shell/web harness;220tasks; blind pairwise comparisons; reasoning max"},{"benchmarkId":"gdpval-aa-2","id":"gdpval-aa-2:meta-muse-1-3-report:QXJ0aWZpY2lhbCBBbmFseXNpcyBTdGlycnVwIHNoZWxsL3dlYiBoYXJuZXNzOzIyMHRhc2tzOyBibGluZCBwYWlyd2lzZSBjb21wYXJpc29uczsgcmVhc29uaW5nIHhoaWdo","name":"Muse Spark1.3 evaluation methodology","notes":"Artificial Analysis Stirrup shell/web harness;220tasks; blind pairwise comparisons; reasoning xhigh"},{"benchmarkId":"jobbench","id":"jobbench:meta-muse-1-3-report:NjV0YXNrczsgbWVhbiBydWJyaWMgc2NvcmU7IG9mZmljaWFsIE9wZW5Db2RlIGhhcm5lc3MgYW5kIGZpbGUtYXdhcmUgZ3JhZGVyOyByZWFzb25pbmcgbWF4","name":"Muse Spark1.3 evaluation methodology","notes":"65tasks; mean rubric score; official OpenCode harness and file-aware grader; reasoning max"},{"benchmarkId":"jobbench","id":"jobbench:meta-muse-1-3-report:NjV0YXNrczsgbWVhbiBydWJyaWMgc2NvcmU7IG9mZmljaWFsIE9wZW5Db2RlIGhhcm5lc3MgYW5kIGZpbGUtYXdhcmUgZ3JhZGVyOyByZWFzb25pbmcgeGhpZ2g","name":"Muse Spark1.3 evaluation methodology","notes":"65tasks; mean rubric score; official OpenCode harness and file-aware grader; reasoning xhigh"},{"benchmarkId":"osworld-partial-2-0-08-08","id":"osworld-partial-2-0-08-08:meta-muse-1-3-report:MTA4dGFza3M7IGNvbW1vbiBpbnRlcm5hbCBHVUkgZnJhbWV3b3JrOyBleGVjdXRpb24tYmFzZWQgY2hlY2tlcnM7IHBhcnRpYWwgbWV0cmljOyByZWFzb25pbmcgbWF4","name":"Muse Spark1.3 evaluation methodology","notes":"108tasks; common internal GUI framework; execution-based checkers; partial metric; reasoning max"},{"benchmarkId":"osworld-partial-2-0-08-08","id":"osworld-partial-2-0-08-08:meta-muse-1-3-report:MTA4dGFza3M7IGNvbW1vbiBpbnRlcm5hbCBHVUkgZnJhbWV3b3JrOyBleGVjdXRpb24tYmFzZWQgY2hlY2tlcnM7IHBhcnRpYWwgbWV0cmljOyByZWFzb25pbmcgeGhpZ2g","name":"Muse Spark1.3 evaluation methodology","notes":"108tasks; common internal GUI framework; execution-based checkers; partial metric; reasoning xhigh"},{"benchmarkId":"osworld-partial-2-0-06-24","id":"osworld-partial-2-0-06-24:meta-muse-1-3-report:MTA4dGFza3M7IGNvbW1vbiBpbnRlcm5hbCBHVUkgZnJhbWV3b3JrOyBleGVjdXRpb24tYmFzZWQgY2hlY2tlcnM7IHBhcnRpYWwgbWV0cmljOyByZWFzb25pbmcgeGhpZ2g","name":"Muse Spark1.3 evaluation methodology","notes":"108tasks; common internal GUI framework; execution-based checkers; partial metric; reasoning xhigh"},{"benchmarkId":"osworld-binary-2-0-08-08","id":"osworld-binary-2-0-08-08:meta-muse-1-3-report:MTA4dGFza3M7IGNvbW1vbiBpbnRlcm5hbCBHVUkgZnJhbWV3b3JrOyBleGVjdXRpb24tYmFzZWQgY2hlY2tlcnM7IGJpbmFyeSBtZXRyaWM7IHJlYXNvbmluZyBtYXg","name":"Muse Spark1.3 evaluation methodology","notes":"108tasks; common internal GUI framework; execution-based checkers; binary metric; reasoning max"},{"benchmarkId":"osworld-binary-2-0-08-08","id":"osworld-binary-2-0-08-08:meta-muse-1-3-report:MTA4dGFza3M7IGNvbW1vbiBpbnRlcm5hbCBHVUkgZnJhbWV3b3JrOyBleGVjdXRpb24tYmFzZWQgY2hlY2tlcnM7IGJpbmFyeSBtZXRyaWM7IHJlYXNvbmluZyB4aGlnaA","name":"Muse Spark1.3 evaluation methodology","notes":"108tasks; common internal GUI framework; execution-based checkers; binary metric; reasoning xhigh"},{"benchmarkId":"osworld-binary-2-0-06-24","id":"osworld-binary-2-0-06-24:meta-muse-1-3-report:MTA4dGFza3M7IGNvbW1vbiBpbnRlcm5hbCBHVUkgZnJhbWV3b3JrOyBleGVjdXRpb24tYmFzZWQgY2hlY2tlcnM7IGJpbmFyeSBtZXRyaWM7IHJlYXNvbmluZyB4aGlnaA","name":"Muse Spark1.3 evaluation methodology","notes":"108tasks; common internal GUI framework; execution-based checkers; binary metric; reasoning xhigh"},{"benchmarkId":"deepsearchqa","id":"deepsearchqa:meta-muse-1-3-report:OTAwcXVlc3Rpb25zOyBjb21tb24gc2VhcmNoIGJhY2tlbmQvYnJvd3NlciBoYXJuZXNzOyBhbnN3ZXItc2V0IEYxOyByZWFzb25pbmcgbWF4","name":"Muse Spark1.3 evaluation methodology","notes":"900questions; common search backend/browser harness; answer-set F1; reasoning max"},{"benchmarkId":"deepsearchqa","id":"deepsearchqa:meta-muse-1-3-report:OTAwcXVlc3Rpb25zOyBjb21tb24gc2VhcmNoIGJhY2tlbmQvYnJvd3NlciBoYXJuZXNzOyBhbnN3ZXItc2V0IEYxOyByZWFzb25pbmcgeGhpZ2g","name":"Muse Spark1.3 evaluation methodology","notes":"900questions; common search backend/browser harness; answer-set F1; reasoning xhigh"},{"benchmarkId":"agentic-if-index-internal","id":"agentic-if-index-internal:meta-muse-1-3-report:SW50ZXJuYWwgY29tcG9zaXRlIGluc3RydWN0aW9uLWZvbGxvd2luZyBldmFsdWF0aW9uczsgbm8gZml4ZWQgdGFzayBjb3VudDsgcmVhc29uaW5nIG1heA","name":"Muse Spark1.3 evaluation methodology","notes":"Internal composite instruction-following evaluations; no fixed task count; reasoning max"},{"benchmarkId":"agentic-if-index-internal","id":"agentic-if-index-internal:meta-muse-1-3-report:SW50ZXJuYWwgY29tcG9zaXRlIGluc3RydWN0aW9uLWZvbGxvd2luZyBldmFsdWF0aW9uczsgbm8gZml4ZWQgdGFzayBjb3VudDsgcmVhc29uaW5nIHhoaWdo","name":"Muse Spark1.3 evaluation methodology","notes":"Internal composite instruction-following evaluations; no fixed task count; reasoning xhigh"},{"benchmarkId":"automationbench-public-v3","id":"automationbench-public-v3:meta-muse-1-3-report:NjAwcHVibGljIHdvcmtmbG93IHRhc2tzOyBkZXRlcm1pbmlzdGljIGVuZC1zdGF0ZSBhc3NlcnRpb25zO3Bhc3NAMTsgcmVhc29uaW5nIG1heA","name":"Muse Spark1.3 evaluation methodology","notes":"600public workflow tasks; deterministic end-state assertions;pass@1; reasoning max"},{"benchmarkId":"automationbench-public-v3","id":"automationbench-public-v3:meta-muse-1-3-report:NjAwcHVibGljIHdvcmtmbG93IHRhc2tzOyBkZXRlcm1pbmlzdGljIGVuZC1zdGF0ZSBhc3NlcnRpb25zO3Bhc3NAMTsgcmVhc29uaW5nIHhoaWdo","name":"Muse Spark1.3 evaluation methodology","notes":"600public workflow tasks; deterministic end-state assertions;pass@1; reasoning xhigh"},{"benchmarkId":"mrcr-2-256k-512k","id":"mrcr-2-256k-512k:meta-muse-1-3-report:OG5lZWRsZTsxMDBleGFtcGxlcy9iYW5kIHJlYmlubmVkIGJ5bzIwMGtfYmFzZTsgbm8gdG9vbHM7IHNlcXVlbmNlLW1hdGNoZXIgcmF0aW87IEdQVCBmcm9tT3BlbkFJY2FyZDsgcmVhc29uaW5nIG1heA","name":"Muse Spark1.3 evaluation methodology","notes":"8needle;100examples/band rebinned byo200k_base; no tools; sequence-matcher ratio; GPT fromOpenAIcard; reasoning max"},{"benchmarkId":"mrcr-2-256k-512k","id":"mrcr-2-256k-512k:meta-muse-1-3-report:OG5lZWRsZTsxMDBleGFtcGxlcy9iYW5kIHJlYmlubmVkIGJ5bzIwMGtfYmFzZTsgbm8gdG9vbHM7IHNlcXVlbmNlLW1hdGNoZXIgcmF0aW87IEdQVCBmcm9tT3BlbkFJY2FyZDsgcmVhc29uaW5nIHhoaWdo","name":"Muse Spark1.3 evaluation methodology","notes":"8needle;100examples/band rebinned byo200k_base; no tools; sequence-matcher ratio; GPT fromOpenAIcard; reasoning xhigh"},{"benchmarkId":"mrcr-2-512k-1m","id":"mrcr-2-512k-1m:meta-muse-1-3-report:OG5lZWRsZTsxMDBleGFtcGxlcy9iYW5kIHJlYmlubmVkIGJ5bzIwMGtfYmFzZTsgbm8gdG9vbHM7IHNlcXVlbmNlLW1hdGNoZXIgcmF0aW87IEdQVCBmcm9tT3BlbkFJY2FyZDsgcmVhc29uaW5nIG1heA","name":"Muse Spark1.3 evaluation methodology","notes":"8needle;100examples/band rebinned byo200k_base; no tools; sequence-matcher ratio; GPT fromOpenAIcard; reasoning max"},{"benchmarkId":"mrcr-2-512k-1m","id":"mrcr-2-512k-1m:meta-muse-1-3-report:OG5lZWRsZTsxMDBleGFtcGxlcy9iYW5kIHJlYmlubmVkIGJ5bzIwMGtfYmFzZTsgbm8gdG9vbHM7IHNlcXVlbmNlLW1hdGNoZXIgcmF0aW87IEdQVCBmcm9tT3BlbkFJY2FyZDsgcmVhc29uaW5nIHhoaWdo","name":"Muse Spark1.3 evaluation methodology","notes":"8needle;100examples/band rebinned byo200k_base; no tools; sequence-matcher ratio; GPT fromOpenAIcard; reasoning xhigh"},{"benchmarkId":"deepswe-1-1-percent","id":"deepswe-1-1-percent:meta-muse-1-3-report:MTEzdGFza3M7IE11c2UxLjNtaW5pLXN3ZS1hZ2VudDsgY29tcGFyYXRvcnMgb2ZmaWNpYWxEYXRhY3VydmUgYm9hcmQ7IHJlYXNvbmluZyBtYXg","name":"Muse Spark1.3 evaluation methodology","notes":"113tasks; Muse1.3mini-swe-agent; comparators officialDatacurve board; reasoning max"},{"benchmarkId":"deepswe-1-1-percent","id":"deepswe-1-1-percent:meta-muse-1-3-report:MTEzdGFza3M7IE11c2UxLjNtaW5pLXN3ZS1hZ2VudDsgY29tcGFyYXRvcnMgb2ZmaWNpYWxEYXRhY3VydmUgYm9hcmQ7IHJlYXNvbmluZyB4aGlnaA","name":"Muse Spark1.3 evaluation methodology","notes":"113tasks; Muse1.3mini-swe-agent; comparators officialDatacurve board; reasoning xhigh"},{"benchmarkId":"swe-atlas-codebase-qna","id":"swe-atlas-codebase-qna:meta-muse-1-3-report:MTI0dGFza3MvMTFyZXBvczsgcHVibGljUW5BIG1pbmktc3dlLWFnZW50IHJ1YnJpYyBwYXNzQDE7IEdQVC9PcHVzIG93bmNhcmRzOyByZWFzb25pbmcgbWF4","name":"Muse Spark1.3 evaluation methodology","notes":"124tasks/11repos; publicQnA mini-swe-agent rubric pass@1; GPT/Opus owncards; reasoning max"},{"benchmarkId":"swe-atlas-codebase-qna","id":"swe-atlas-codebase-qna:meta-muse-1-3-report:MTI0dGFza3MvMTFyZXBvczsgcHVibGljUW5BIG1pbmktc3dlLWFnZW50IHJ1YnJpYyBwYXNzQDE7IEdQVC9PcHVzIG93bmNhcmRzOyByZWFzb25pbmcgeGhpZ2g","name":"Muse Spark1.3 evaluation methodology","notes":"124tasks/11repos; publicQnA mini-swe-agent rubric pass@1; GPT/Opus owncards; reasoning xhigh"},{"benchmarkId":"terminal-bench-2-1-percent-higher","id":"terminal-bench-2-1-percent-higher:meta-muse-1-3-report:ODl0YXNrczsgZWFjaCBtb2RlbCBuYXRpdmVjb2RpbmcgaGFybmVzcyBpbiBpbnRlcm5hbGZyYW1ld29yay9jbG91ZHNhbmRib3g7IG9mZmljaWFsdmVyaWZpZXI7IEdQVCBvd25jYXJkOyByZWFzb25pbmcgbWF4OyBuYXRpdmUgaGFybmVzcyBmYW1pbHk6IE1ldGEgKGV4YWN0IGhhcm5lc3MgcmV2aXNpb24gbm90IHNwZWNpZmllZCk","name":"Muse Spark1.3 evaluation methodology","notes":"89tasks; each model nativecoding harness in internalframework/cloudsandbox; officialverifier; GPT owncard; reasoning max; native harness family: Meta (exact harness revision not specified)"},{"benchmarkId":"terminal-bench-2-1-percent-higher","id":"terminal-bench-2-1-percent-higher:meta-muse-1-3-report:ODl0YXNrczsgZWFjaCBtb2RlbCBuYXRpdmVjb2RpbmcgaGFybmVzcyBpbiBpbnRlcm5hbGZyYW1ld29yay9jbG91ZHNhbmRib3g7IG9mZmljaWFsdmVyaWZpZXI7IEdQVCBvd25jYXJkOyByZWFzb25pbmcgeGhpZ2g7IG5hdGl2ZSBoYXJuZXNzIGZhbWlseTogTWV0YSAoZXhhY3QgaGFybmVzcyByZXZpc2lvbiBub3Qgc3BlY2lmaWVkKQ","name":"Muse Spark1.3 evaluation methodology","notes":"89tasks; each model nativecoding harness in internalframework/cloudsandbox; officialverifier; GPT owncard; reasoning xhigh; native harness family: Meta (exact harness revision not specified)"},{"benchmarkId":"terminal-bench-2-1-percent-higher","id":"terminal-bench-2-1-percent-higher:meta-muse-1-3-report:ODl0YXNrczsgZWFjaCBtb2RlbCBuYXRpdmVjb2RpbmcgaGFybmVzcyBpbiBpbnRlcm5hbGZyYW1ld29yay9jbG91ZHNhbmRib3g7IG9mZmljaWFsdmVyaWZpZXI7IEdQVCBvd25jYXJkOyByZWFzb25pbmcgbWF4OyBuYXRpdmUgaGFybmVzcyBmYW1pbHk6IE9wZW5BSSAoZXhhY3QgaGFybmVzcyByZXZpc2lvbiBub3Qgc3BlY2lmaWVkKQ","name":"Muse Spark1.3 evaluation methodology","notes":"89tasks; each model nativecoding harness in internalframework/cloudsandbox; officialverifier; GPT owncard; reasoning max; native harness family: OpenAI (exact harness revision not specified)"},{"benchmarkId":"terminal-bench-2-1-percent-higher","id":"terminal-bench-2-1-percent-higher:meta-muse-1-3-report:ODl0YXNrczsgZWFjaCBtb2RlbCBuYXRpdmVjb2RpbmcgaGFybmVzcyBpbiBpbnRlcm5hbGZyYW1ld29yay9jbG91ZHNhbmRib3g7IG9mZmljaWFsdmVyaWZpZXI7IEdQVCBvd25jYXJkOyByZWFzb25pbmcgbWF4OyBuYXRpdmUgaGFybmVzcyBmYW1pbHk6IEFudGhyb3BpYyAoZXhhY3QgaGFybmVzcyByZXZpc2lvbiBub3Qgc3BlY2lmaWVkKQ","name":"Muse Spark1.3 evaluation methodology","notes":"89tasks; each model nativecoding harness in internalframework/cloudsandbox; officialverifier; GPT owncard; reasoning max; native harness family: Anthropic (exact harness revision not specified)"},{"benchmarkId":"deepswe-1-1-percent","id":"deepswe-1-1-percent:google-gemini-3-8-card:RGF0YWN1cnZlIGhpZ2hlc3QgcmVwb3J0ZWQgZWZmb3J0OyBHZW1pbmkzLjhzZWxmY29tcHV0ZWQgbWluaS1zd2UtYWdlbnQgaGlnaHRoaW5raW5n","name":"Gemini3.8Flash Model Card","notes":"Datacurve highest reported effort; Gemini3.8selfcomputed mini-swe-agent highthinking"},{"benchmarkId":"gdpval-aa-2","id":"gdpval-aa-2:google-gemini-3-8-card:QXJ0aWZpY2lhbCBBbmFseXNpcyBwdWJsaWNib2FyZCBzbmFwc2hvdDsgZWZmb3J0IGFzIHJlcG9ydGVk","name":"Gemini3.8Flash Model Card","notes":"Artificial Analysis publicboard snapshot; effort as reported"},{"benchmarkId":"terminal-bench-2-1-percent-higher","id":"terminal-bench-2-1-percent-higher:google-gemini-3-8-card:VGVybWludXMyb25seTsgR2VtaW5pIHNlbGZjb21wdXRlZCwgb3RoZXJtb2RlbHMgb2ZmaWNpYWxib2FyZC9BcnRpZmljaWFsQW5hbHlzaXM","name":"Gemini3.8Flash Model Card","notes":"Terminus2only; Gemini selfcomputed, othermodels officialboard/ArtificialAnalysis"},{"benchmarkId":"terminal-bench-4-0-percent","id":"terminal-bench-4-0-percent:google-gemini-3-8-card:T2ZmaWNpYWxwdWJsaWNib2FyZCBoaWdoZXN0IHNjb3JpbmcgdGhpbmtpbmcgbGV2ZWw7IG5hdGl2ZWFnZW50cyBtYXkgZGlmZmVy","name":"Gemini3.8Flash Model Card","notes":"Officialpublicboard highest scoring thinking level; nativeagents may differ"},{"benchmarkId":"gdp-pdf","id":"gdp-pdf:google-gemini-3-8-card:QWxsLXBhc3MgcmF0ZTsgYWxsbW9kZWxzIHNlbGZjb21wdXRlZCBieUdvb2dsZQ","name":"Gemini3.8Flash Model Card","notes":"All-pass rate; allmodels selfcomputed byGoogle"},{"benchmarkId":"charxiv-reasoning","id":"charxiv-reasoning:google-gemini-3-8-card:Tm8gdG9vbHM7IEdlbWluaS9HUFQvT3B1cyBzZWxmY29tcHV0ZWQ7IFNvbm5ldCBzZWxmcmVwb3J0ZWQ","name":"Gemini3.8Flash Model Card","notes":"No tools; Gemini/GPT/Opus selfcomputed; Sonnet selfreported"},{"benchmarkId":"lvbench-static","id":"lvbench-static:google-gemini-3-8-card:Tm8gdG9vbHM7MTAyNGZyYW1lcyBHZW1pbmkvR1BULDMwMGZyYW1lcyBDbGF1ZGUgZHVlQVBJbGltaXQ7IG1vZGVsLXNwZWNpZmljIGZyYW1lIGJ1ZGdldDogMTAyNA","name":"Gemini3.8Flash Model Card","notes":"No tools;1024frames Gemini/GPT,300frames Claude dueAPIlimit; model-specific frame budget: 1024"},{"benchmarkId":"lvbench-static","id":"lvbench-static:google-gemini-3-8-card:Tm8gdG9vbHM7MTAyNGZyYW1lcyBHZW1pbmkvR1BULDMwMGZyYW1lcyBDbGF1ZGUgZHVlQVBJbGltaXQ7IG1vZGVsLXNwZWNpZmljIGZyYW1lIGJ1ZGdldDogMzAw","name":"Gemini3.8Flash Model Card","notes":"No tools;1024frames Gemini/GPT,300frames Claude dueAPIlimit; model-specific frame budget: 300"},{"benchmarkId":"lvbench-agentic","id":"lvbench-agentic:google-gemini-3-8-card:Q2FyZCBsYWJlbHMgYWdlbnRpYzsgbGlua2VkbWV0aG9kb2xvZ3kgZGVzY3JpYmVzIG9ubHkgbm8tdG9vbHMgc3RhdGljIHNldHVw","name":"Gemini3.8Flash Model Card","notes":"Card labels agentic; linkedmethodology describes only no-tools static setup"},{"benchmarkId":"osworld-partial-2-0-pre-08-08","id":"osworld-partial-2-0-pre-08-08:google-gemini-3-8-card:UGFydGlhbHNjb3JlOyBiYXRjaHRvb2xzOzEwODBwLzUwMHN0ZXBzOyBHZW1pbmkvU29ubmV0IGJlc3RvZjNydW5zOyBzY3JlZW5zaG90b25seTsgb2ZmaWNpYWxDVUFoYXJuZXNz","name":"Gemini3.8Flash Model Card","notes":"Partialscore; batchtools;1080p/500steps; Gemini/Sonnet bestof3runs; screenshotonly; officialCUAharness"},{"benchmarkId":"osworld-partial-2-0-august2026-fixed-tasks","id":"osworld-partial-2-0-august2026-fixed-tasks:google-gemini-3-8-card:UGFydGlhbHNjb3JlOyBiYXRjaHRvb2xzOzEwODBwLzUwMHN0ZXBzOyBHZW1pbmkvU29ubmV0IGJlc3RvZjNydW5zOyBzY3JlZW5zaG90b25seTsgb2ZmaWNpYWxDVUFoYXJuZXNz","name":"Gemini3.8Flash Model Card","notes":"Partialscore; batchtools;1080p/500steps; Gemini/Sonnet bestof3runs; screenshotonly; officialCUAharness"},{"benchmarkId":"osworld-partial-2-0-task-revision-unverified-gpt-5-6-sol","id":"osworld-partial-2-0-task-revision-unverified-gpt-5-6-sol:google-gemini-3-8-card:UGFydGlhbHNjb3JlOyBiYXRjaHRvb2xzOzEwODBwLzUwMHN0ZXBzOyBHZW1pbmkvU29ubmV0IGJlc3RvZjNydW5zOyBzY3JlZW5zaG90b25seTsgb2ZmaWNpYWxDVUFoYXJuZXNz","name":"Gemini3.8Flash Model Card","notes":"Partialscore; batchtools;1080p/500steps; Gemini/Sonnet bestof3runs; screenshotonly; officialCUAharness"},{"benchmarkId":"osworld-partial-2-0-task-revision-unverified-gpt-5-6-terra","id":"osworld-partial-2-0-task-revision-unverified-gpt-5-6-terra:google-gemini-3-8-card:UGFydGlhbHNjb3JlOyBiYXRjaHRvb2xzOzEwODBwLzUwMHN0ZXBzOyBHZW1pbmkvU29ubmV0IGJlc3RvZjNydW5zOyBzY3JlZW5zaG90b25seTsgb2ZmaWNpYWxDVUFoYXJuZXNz","name":"Gemini3.8Flash Model Card","notes":"Partialscore; batchtools;1080p/500steps; Gemini/Sonnet bestof3runs; screenshotonly; officialCUAharness"},{"benchmarkId":"biomysterybench-human-solvable","id":"biomysterybench-human-solvable:google-gemini-3-8-card:TGludXh0ZXJtaW5hbCxiaW9pbmZvdG9vbHMsUHl0aG9uLFI7IGFsbG93bGlzdGVkbmV0d29yazsgR2VtaW5pL0dQVHNlbGZjb21wdXRlZCxDbGF1ZGVwcm92aWRlcnJlcG9ydHM","name":"Gemini3.8Flash Model Card","notes":"Linuxterminal,bioinfotools,Python,R; allowlistednetwork; Gemini/GPTselfcomputed,Claudeproviderreports"},{"benchmarkId":"biomysterybench-human-difficult","id":"biomysterybench-human-difficult:google-gemini-3-8-card:TGludXh0ZXJtaW5hbCxiaW9pbmZvdG9vbHMsUHl0aG9uLFI7IGFsbG93bGlzdGVkbmV0d29yazsgR2VtaW5pL0dQVHNlbGZjb21wdXRlZCxDbGF1ZGVwcm92aWRlcnJlcG9ydHM","name":"Gemini3.8Flash Model Card","notes":"Linuxterminal,bioinfotools,Python,R; allowlistednetwork; Gemini/GPTselfcomputed,Claudeproviderreports"},{"benchmarkId":"labbench-2","id":"labbench-2:google-gemini-3-8-card:U2VsZmNvbXB1dGVkOyBMaW51eHRlcm1pbmFsLGJpb2luZm90b29scyxQeXRob24sUixuZXR3b3JrOyBtYWNyb2F2ZXJhZ2UxMXN1YnRhc2tz","name":"Gemini3.8Flash Model Card","notes":"Selfcomputed; Linuxterminal,bioinfotools,Python,R,network; macroaverage11subtasks"},{"benchmarkId":"hle-full-w-tools","id":"hle-full-w-tools:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEhMRS1GdWxsICh3LyB0b29scykuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMu","name":"Kimi K2.6 official model card","notes":"Kimi K2.6 card, HLE-Full (w/ tools). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"benchmarkId":"browsecomp-percent","id":"browsecomp-percent:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEJyb3dzZUNvbXAuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMu","name":"Kimi K2.6 official model card","notes":"Kimi K2.6 card, BrowseComp. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"benchmarkId":"browsecomp-agent-swarm","id":"browsecomp-agent-swarm:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEJyb3dzZUNvbXAgKEFnZW50IFN3YXJtKS4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4","name":"Kimi K2.6 official model card","notes":"Kimi K2.6 card, BrowseComp (Agent Swarm). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"benchmarkId":"deepsearchqa-f1-score","id":"deepsearchqa-f1-score:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIERlZXBTZWFyY2hRQSAoZjEtc2NvcmUpLiBUaGlua2luZyBtb2RlOyB0ZW1wZXJhdHVyZSAxLjAsIHRvcC1wIDEuMCwgY29udGV4dCAyNjIxNDQuIENvbXBhcmF0b3Igc2V0dGluZ3M6IEdQVC01LjQgeGhpZ2gsIE9wdXM0LjYgbWF4LCBHZW1pbmkzLjFQcm8gaGlnaC4gU2VlIGJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMuIE9ubHkgb3duIEsyLjYgYW5kIHN0YXJyZWQgcmUtZXZhbHVhdGlvbnMgY2FuIGZvcm0gbmV3IGNvbXBhcmlzb25zLg","name":"Kimi K2.6 official model card","notes":"Kimi K2.6 card, DeepSearchQA (f1-score). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"benchmarkId":"deepsearchqa-accuracy","id":"deepsearchqa-accuracy:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIERlZXBTZWFyY2hRQSAoYWNjdXJhY3kpLiBUaGlua2luZyBtb2RlOyB0ZW1wZXJhdHVyZSAxLjAsIHRvcC1wIDEuMCwgY29udGV4dCAyNjIxNDQuIENvbXBhcmF0b3Igc2V0dGluZ3M6IEdQVC01LjQgeGhpZ2gsIE9wdXM0LjYgbWF4LCBHZW1pbmkzLjFQcm8gaGlnaC4gU2VlIGJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMuIE9ubHkgb3duIEsyLjYgYW5kIHN0YXJyZWQgcmUtZXZhbHVhdGlvbnMgY2FuIGZvcm0gbmV3IGNvbXBhcmlzb25zLg","name":"Kimi K2.6 official model card","notes":"Kimi K2.6 card, DeepSearchQA (accuracy). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"benchmarkId":"widesearch-item-f1","id":"widesearch-item-f1:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIFdpZGVTZWFyY2ggKGl0ZW0tZjEpLiBUaGlua2luZyBtb2RlOyB0ZW1wZXJhdHVyZSAxLjAsIHRvcC1wIDEuMCwgY29udGV4dCAyNjIxNDQuIENvbXBhcmF0b3Igc2V0dGluZ3M6IEdQVC01LjQgeGhpZ2gsIE9wdXM0LjYgbWF4LCBHZW1pbmkzLjFQcm8gaGlnaC4gU2VlIGJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMuIE9ubHkgb3duIEsyLjYgYW5kIHN0YXJyZWQgcmUtZXZhbHVhdGlvbnMgY2FuIGZvcm0gbmV3IGNvbXBhcmlzb25zLg","name":"Kimi K2.6 official model card","notes":"Kimi K2.6 card, WideSearch (item-f1). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"benchmarkId":"toolathlon","id":"toolathlon:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIFRvb2xhdGhsb24uIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMu","name":"Kimi K2.6 official model card","notes":"Kimi K2.6 card, Toolathlon. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"benchmarkId":"mcpmark","id":"mcpmark:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIE1DUE1hcmsuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMu","name":"Kimi K2.6 official model card","notes":"Kimi K2.6 card, MCPMark. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"benchmarkId":"claw-eval-pass-3-percent","id":"claw-eval-pass-3-percent:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIENsYXcgRXZhbCAocGFzc14zKS4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4","name":"Kimi K2.6 official model card","notes":"Kimi K2.6 card, Claw Eval (pass^3). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"benchmarkId":"claw-eval-pass-3","id":"claw-eval-pass-3:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIENsYXcgRXZhbCAocGFzc0AzKS4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4","name":"Kimi K2.6 official model card","notes":"Kimi K2.6 card, Claw Eval (pass@3). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"benchmarkId":"apex-agents","id":"apex-agents:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEFQRVgtQWdlbnRzLiBUaGlua2luZyBtb2RlOyB0ZW1wZXJhdHVyZSAxLjAsIHRvcC1wIDEuMCwgY29udGV4dCAyNjIxNDQuIENvbXBhcmF0b3Igc2V0dGluZ3M6IEdQVC01LjQgeGhpZ2gsIE9wdXM0LjYgbWF4LCBHZW1pbmkzLjFQcm8gaGlnaC4gU2VlIGJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMuIE9ubHkgb3duIEsyLjYgYW5kIHN0YXJyZWQgcmUtZXZhbHVhdGlvbnMgY2FuIGZvcm0gbmV3IGNvbXBhcmlzb25zLg","name":"Kimi K2.6 official model card","notes":"Kimi K2.6 card, APEX-Agents. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"benchmarkId":"osworld-verified-percent","id":"osworld-verified-percent:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIE9TV29ybGQtVmVyaWZpZWQuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMu","name":"Kimi K2.6 official model card","notes":"Kimi K2.6 card, OSWorld-Verified. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"benchmarkId":"terminal-bench-2-0-terminus-2","id":"terminal-bench-2-0-terminus-2:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIFRlcm1pbmFsLUJlbmNoIDIuMCAoVGVybWludXMtMikuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMuIENvZGluZzogYXZlcmFnZSBvZiB0ZW4gcnVuczsgVGVybWluYWwgdXNlcyBUZXJtaW51czIgcHJlc2VydmUtdGhpbmtpbmc7IFNXRSB1c2VzIGluLWhvdXNlIFNXRS1hZ2VudC4","name":"Kimi K2.6 official model card","notes":"Kimi K2.6 card, Terminal-Bench 2.0 (Terminus-2). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent."},{"benchmarkId":"swe-bench-pro-percent","id":"swe-bench-pro-percent:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIFNXRS1CZW5jaCBQcm8uIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMuIENvZGluZzogYXZlcmFnZSBvZiB0ZW4gcnVuczsgVGVybWluYWwgdXNlcyBUZXJtaW51czIgcHJlc2VydmUtdGhpbmtpbmc7IFNXRSB1c2VzIGluLWhvdXNlIFNXRS1hZ2VudC4","name":"Kimi K2.6 official model card","notes":"Kimi K2.6 card, SWE-Bench Pro. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent."},{"benchmarkId":"swe-bench-multilingual","id":"swe-bench-multilingual:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIFNXRS1CZW5jaCBNdWx0aWxpbmd1YWwuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMuIENvZGluZzogYXZlcmFnZSBvZiB0ZW4gcnVuczsgVGVybWluYWwgdXNlcyBUZXJtaW51czIgcHJlc2VydmUtdGhpbmtpbmc7IFNXRSB1c2VzIGluLWhvdXNlIFNXRS1hZ2VudC4","name":"Kimi K2.6 official model card","notes":"Kimi K2.6 card, SWE-Bench Multilingual. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent."},{"benchmarkId":"swe-bench-verified-percent","id":"swe-bench-verified-percent:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIFNXRS1CZW5jaCBWZXJpZmllZC4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4gQ29kaW5nOiBhdmVyYWdlIG9mIHRlbiBydW5zOyBUZXJtaW5hbCB1c2VzIFRlcm1pbnVzMiBwcmVzZXJ2ZS10aGlua2luZzsgU1dFIHVzZXMgaW4taG91c2UgU1dFLWFnZW50Lg","name":"Kimi K2.6 official model card","notes":"Kimi K2.6 card, SWE-Bench Verified. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent."},{"benchmarkId":"scicode","id":"scicode:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIFNjaUNvZGUuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMuIENvZGluZzogYXZlcmFnZSBvZiB0ZW4gcnVuczsgVGVybWluYWwgdXNlcyBUZXJtaW51czIgcHJlc2VydmUtdGhpbmtpbmc7IFNXRSB1c2VzIGluLWhvdXNlIFNXRS1hZ2VudC4","name":"Kimi K2.6 official model card","notes":"Kimi K2.6 card, SciCode. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent."},{"benchmarkId":"ojbench-python","id":"ojbench-python:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIE9KQmVuY2ggKHB5dGhvbikuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMuIENvZGluZzogYXZlcmFnZSBvZiB0ZW4gcnVuczsgVGVybWluYWwgdXNlcyBUZXJtaW51czIgcHJlc2VydmUtdGhpbmtpbmc7IFNXRSB1c2VzIGluLWhvdXNlIFNXRS1hZ2VudC4","name":"Kimi K2.6 official model card","notes":"Kimi K2.6 card, OJBench (python). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent."},{"benchmarkId":"livecodebench-v6","id":"livecodebench-v6:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIExpdmVDb2RlQmVuY2ggKHY2KS4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4gQ29kaW5nOiBhdmVyYWdlIG9mIHRlbiBydW5zOyBUZXJtaW5hbCB1c2VzIFRlcm1pbnVzMiBwcmVzZXJ2ZS10aGlua2luZzsgU1dFIHVzZXMgaW4taG91c2UgU1dFLWFnZW50Lg","name":"Kimi K2.6 official model card","notes":"Kimi K2.6 card, LiveCodeBench (v6). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent."},{"benchmarkId":"hle-full","id":"hle-full:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEhMRS1GdWxsLiBUaGlua2luZyBtb2RlOyB0ZW1wZXJhdHVyZSAxLjAsIHRvcC1wIDEuMCwgY29udGV4dCAyNjIxNDQuIENvbXBhcmF0b3Igc2V0dGluZ3M6IEdQVC01LjQgeGhpZ2gsIE9wdXM0LjYgbWF4LCBHZW1pbmkzLjFQcm8gaGlnaC4gU2VlIGJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMuIE9ubHkgb3duIEsyLjYgYW5kIHN0YXJyZWQgcmUtZXZhbHVhdGlvbnMgY2FuIGZvcm0gbmV3IGNvbXBhcmlzb25zLiBSZWFzb25pbmc6IDk4MzA0IGdlbmVyYXRpb24gdG9rZW5zOyBITEUgZnVsbCBzZXQu","name":"Kimi K2.6 official model card","notes":"Kimi K2.6 card, HLE-Full. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Reasoning: 98304 generation tokens; HLE full set."},{"benchmarkId":"aime-2026","id":"aime-2026:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEFJTUUgMjAyNi4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4gUmVhc29uaW5nOiA5ODMwNCBnZW5lcmF0aW9uIHRva2VuczsgSExFIGZ1bGwgc2V0Lg","name":"Kimi K2.6 official model card","notes":"Kimi K2.6 card, AIME 2026. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Reasoning: 98304 generation tokens; HLE full set."},{"benchmarkId":"hmmt-2026-feb","id":"hmmt-2026-feb:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEhNTVQgMjAyNiAoRmViKS4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4gUmVhc29uaW5nOiA5ODMwNCBnZW5lcmF0aW9uIHRva2VuczsgSExFIGZ1bGwgc2V0Lg","name":"Kimi K2.6 official model card","notes":"Kimi K2.6 card, HMMT 2026 (Feb). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Reasoning: 98304 generation tokens; HLE full set."},{"benchmarkId":"imo-answerbench","id":"imo-answerbench:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIElNTy1BbnN3ZXJCZW5jaC4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4gUmVhc29uaW5nOiA5ODMwNCBnZW5lcmF0aW9uIHRva2VuczsgSExFIGZ1bGwgc2V0Lg","name":"Kimi K2.6 official model card","notes":"Kimi K2.6 card, IMO-AnswerBench. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Reasoning: 98304 generation tokens; HLE full set."},{"benchmarkId":"gpqa-diamond-percent","id":"gpqa-diamond-percent:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEdQUUEtRGlhbW9uZC4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4gUmVhc29uaW5nOiA5ODMwNCBnZW5lcmF0aW9uIHRva2VuczsgSExFIGZ1bGwgc2V0Lg","name":"Kimi K2.6 official model card","notes":"Kimi K2.6 card, GPQA-Diamond. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Reasoning: 98304 generation tokens; HLE full set."},{"benchmarkId":"mmmu-pro","id":"mmmu-pro:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIE1NTVUtUHJvLiBUaGlua2luZyBtb2RlOyB0ZW1wZXJhdHVyZSAxLjAsIHRvcC1wIDEuMCwgY29udGV4dCAyNjIxNDQuIENvbXBhcmF0b3Igc2V0dGluZ3M6IEdQVC01LjQgeGhpZ2gsIE9wdXM0LjYgbWF4LCBHZW1pbmkzLjFQcm8gaGlnaC4gU2VlIGJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMuIE9ubHkgb3duIEsyLjYgYW5kIHN0YXJyZWQgcmUtZXZhbHVhdGlvbnMgY2FuIGZvcm0gbmV3IGNvbXBhcmlzb25zLiBWaXNpb246IDk4MzA0IHRva2VucywgYXZlcmFnZSBvZiB0aHJlZSBydW5zOyBQeXRob24gY29uZGl0aW9uIHVzZXMgNjU1MzYgdG9rZW5zIHBlciBzdGVwLCBhdCBtb3N0NTBzdGVwcy4","name":"Kimi K2.6 official model card","notes":"Kimi K2.6 card, MMMU-Pro. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps."},{"benchmarkId":"mmmu-pro-w-python","id":"mmmu-pro-w-python:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIE1NTVUtUHJvICh3LyBweXRob24pLiBUaGlua2luZyBtb2RlOyB0ZW1wZXJhdHVyZSAxLjAsIHRvcC1wIDEuMCwgY29udGV4dCAyNjIxNDQuIENvbXBhcmF0b3Igc2V0dGluZ3M6IEdQVC01LjQgeGhpZ2gsIE9wdXM0LjYgbWF4LCBHZW1pbmkzLjFQcm8gaGlnaC4gU2VlIGJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMuIE9ubHkgb3duIEsyLjYgYW5kIHN0YXJyZWQgcmUtZXZhbHVhdGlvbnMgY2FuIGZvcm0gbmV3IGNvbXBhcmlzb25zLiBWaXNpb246IDk4MzA0IHRva2VucywgYXZlcmFnZSBvZiB0aHJlZSBydW5zOyBQeXRob24gY29uZGl0aW9uIHVzZXMgNjU1MzYgdG9rZW5zIHBlciBzdGVwLCBhdCBtb3N0NTBzdGVwcy4","name":"Kimi K2.6 official model card","notes":"Kimi K2.6 card, MMMU-Pro (w/ python). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps."},{"benchmarkId":"charxiv-rq","id":"charxiv-rq:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIENoYXJYaXYgKFJRKS4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4gVmlzaW9uOiA5ODMwNCB0b2tlbnMsIGF2ZXJhZ2Ugb2YgdGhyZWUgcnVuczsgUHl0aG9uIGNvbmRpdGlvbiB1c2VzIDY1NTM2IHRva2VucyBwZXIgc3RlcCwgYXQgbW9zdDUwc3RlcHMu","name":"Kimi K2.6 official model card","notes":"Kimi K2.6 card, CharXiv (RQ). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps."},{"benchmarkId":"charxiv-rq-w-python","id":"charxiv-rq-w-python:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIENoYXJYaXYgKFJRKSAody8gcHl0aG9uKS4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4gVmlzaW9uOiA5ODMwNCB0b2tlbnMsIGF2ZXJhZ2Ugb2YgdGhyZWUgcnVuczsgUHl0aG9uIGNvbmRpdGlvbiB1c2VzIDY1NTM2IHRva2VucyBwZXIgc3RlcCwgYXQgbW9zdDUwc3RlcHMu","name":"Kimi K2.6 official model card","notes":"Kimi K2.6 card, CharXiv (RQ) (w/ python). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps."},{"benchmarkId":"mathvision","id":"mathvision:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIE1hdGhWaXNpb24uIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMuIFZpc2lvbjogOTgzMDQgdG9rZW5zLCBhdmVyYWdlIG9mIHRocmVlIHJ1bnM7IFB5dGhvbiBjb25kaXRpb24gdXNlcyA2NTUzNiB0b2tlbnMgcGVyIHN0ZXAsIGF0IG1vc3Q1MHN0ZXBzLg","name":"Kimi K2.6 official model card","notes":"Kimi K2.6 card, MathVision. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps."},{"benchmarkId":"mathvision-w-python","id":"mathvision-w-python:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIE1hdGhWaXNpb24gKHcvIHB5dGhvbikuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMuIFZpc2lvbjogOTgzMDQgdG9rZW5zLCBhdmVyYWdlIG9mIHRocmVlIHJ1bnM7IFB5dGhvbiBjb25kaXRpb24gdXNlcyA2NTUzNiB0b2tlbnMgcGVyIHN0ZXAsIGF0IG1vc3Q1MHN0ZXBzLg","name":"Kimi K2.6 official model card","notes":"Kimi K2.6 card, MathVision (w/ python). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps."},{"benchmarkId":"babyvision","id":"babyvision:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEJhYnlWaXNpb24uIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMuIFZpc2lvbjogOTgzMDQgdG9rZW5zLCBhdmVyYWdlIG9mIHRocmVlIHJ1bnM7IFB5dGhvbiBjb25kaXRpb24gdXNlcyA2NTUzNiB0b2tlbnMgcGVyIHN0ZXAsIGF0IG1vc3Q1MHN0ZXBzLg","name":"Kimi K2.6 official model card","notes":"Kimi K2.6 card, BabyVision. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps."},{"benchmarkId":"babyvision-w-python","id":"babyvision-w-python:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEJhYnlWaXNpb24gKHcvIHB5dGhvbikuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMuIFZpc2lvbjogOTgzMDQgdG9rZW5zLCBhdmVyYWdlIG9mIHRocmVlIHJ1bnM7IFB5dGhvbiBjb25kaXRpb24gdXNlcyA2NTUzNiB0b2tlbnMgcGVyIHN0ZXAsIGF0IG1vc3Q1MHN0ZXBzLg","name":"Kimi K2.6 official model card","notes":"Kimi K2.6 card, BabyVision (w/ python). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps."},{"benchmarkId":"v-w-python","id":"v-w-python:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIFYqICh3LyBweXRob24pLiBUaGlua2luZyBtb2RlOyB0ZW1wZXJhdHVyZSAxLjAsIHRvcC1wIDEuMCwgY29udGV4dCAyNjIxNDQuIENvbXBhcmF0b3Igc2V0dGluZ3M6IEdQVC01LjQgeGhpZ2gsIE9wdXM0LjYgbWF4LCBHZW1pbmkzLjFQcm8gaGlnaC4gU2VlIGJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMuIE9ubHkgb3duIEsyLjYgYW5kIHN0YXJyZWQgcmUtZXZhbHVhdGlvbnMgY2FuIGZvcm0gbmV3IGNvbXBhcmlzb25zLiBWaXNpb246IDk4MzA0IHRva2VucywgYXZlcmFnZSBvZiB0aHJlZSBydW5zOyBQeXRob24gY29uZGl0aW9uIHVzZXMgNjU1MzYgdG9rZW5zIHBlciBzdGVwLCBhdCBtb3N0NTBzdGVwcy4","name":"Kimi K2.6 official model card","notes":"Kimi K2.6 card, V* (w/ python). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps."},{"benchmarkId":"cursorbench-4-0","id":"cursorbench-4-0:xai-grok47:R3JvayA0LjcgWEhpZ2ggKERlZXBTV0U6IEhpZ2gpOyBHcm9rIDQuNiBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IENsYXVkZSBGYWJsZSA1LjEgTWF4LiBMYXVuY2ggY29tcGFyaXNvbjsgcGVyLW1vZGVsIGFnZW50IGhhcm5lc3MsIGV2YWx1YXRpb24gYnVkZ2V0IGFuZCByZXJ1biBwcm92ZW5hbmNlIGFyZSBub3Qgc3BlY2lmaWVkLg","name":"Grok 4.7 launch comparison","notes":"Grok 4.7 XHigh (DeepSWE: High); Grok 4.6 High; GPT-5.6 Sol Max; Claude Fable 5.1 Max. Launch comparison; per-model agent harness, evaluation budget and rerun provenance are not specified."},{"benchmarkId":"deepswe-v1-1-1-1","id":"deepswe-v1-1-1-1:xai-grok47:R3JvayA0LjcgWEhpZ2ggKERlZXBTV0U6IEhpZ2gpOyBHcm9rIDQuNiBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IENsYXVkZSBGYWJsZSA1LjEgTWF4LiBMYXVuY2ggY29tcGFyaXNvbjsgcGVyLW1vZGVsIGFnZW50IGhhcm5lc3MsIGV2YWx1YXRpb24gYnVkZ2V0IGFuZCByZXJ1biBwcm92ZW5hbmNlIGFyZSBub3Qgc3BlY2lmaWVkLg","name":"Grok 4.7 launch comparison","notes":"Grok 4.7 XHigh (DeepSWE: High); Grok 4.6 High; GPT-5.6 Sol Max; Claude Fable 5.1 Max. Launch comparison; per-model agent harness, evaluation budget and rerun provenance are not specified."},{"benchmarkId":"eebench","id":"eebench:xai-grok47:R3JvayA0LjcgWEhpZ2ggKERlZXBTV0U6IEhpZ2gpOyBHcm9rIDQuNiBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IENsYXVkZSBGYWJsZSA1LjEgTWF4LiBMYXVuY2ggY29tcGFyaXNvbjsgcGVyLW1vZGVsIGFnZW50IGhhcm5lc3MsIGV2YWx1YXRpb24gYnVkZ2V0IGFuZCByZXJ1biBwcm92ZW5hbmNlIGFyZSBub3Qgc3BlY2lmaWVkLg","name":"Grok 4.7 launch comparison","notes":"Grok 4.7 XHigh (DeepSWE: High); Grok 4.6 High; GPT-5.6 Sol Max; Claude Fable 5.1 Max. Launch comparison; per-model agent harness, evaluation budget and rerun provenance are not specified."},{"benchmarkId":"aa-briefcase-1-1","id":"aa-briefcase-1-1:xai-grok47:R3JvayA0LjcgWEhpZ2ggKERlZXBTV0U6IEhpZ2gpOyBHcm9rIDQuNiBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IENsYXVkZSBGYWJsZSA1LjEgTWF4LiBMYXVuY2ggY29tcGFyaXNvbjsgcGVyLW1vZGVsIGFnZW50IGhhcm5lc3MsIGV2YWx1YXRpb24gYnVkZ2V0IGFuZCByZXJ1biBwcm92ZW5hbmNlIGFyZSBub3Qgc3BlY2lmaWVkLg","name":"Grok 4.7 launch comparison","notes":"Grok 4.7 XHigh (DeepSWE: High); Grok 4.6 High; GPT-5.6 Sol Max; Claude Fable 5.1 Max. Launch comparison; per-model agent harness, evaluation budget and rerun provenance are not specified."},{"benchmarkId":"terminal-bench-4-0-pct","id":"terminal-bench-4-0-pct:xai-grok47:R3JvayA0LjcgWEhpZ2ggKERlZXBTV0U6IEhpZ2gpOyBHcm9rIDQuNiBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IENsYXVkZSBGYWJsZSA1LjEgTWF4LiBMYXVuY2ggY29tcGFyaXNvbjsgcGVyLW1vZGVsIGFnZW50IGhhcm5lc3MsIGV2YWx1YXRpb24gYnVkZ2V0IGFuZCByZXJ1biBwcm92ZW5hbmNlIGFyZSBub3Qgc3BlY2lmaWVkLg","name":"Grok 4.7 launch comparison","notes":"Grok 4.7 XHigh (DeepSWE: High); Grok 4.6 High; GPT-5.6 Sol Max; Claude Fable 5.1 Max. Launch comparison; per-model agent harness, evaluation budget and rerun provenance are not specified."},{"benchmarkId":"harvey-legal-agent-benchmark","id":"harvey-legal-agent-benchmark:xai-grok47:R3JvayA0LjcgWEhpZ2ggKERlZXBTV0U6IEhpZ2gpOyBHcm9rIDQuNiBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IENsYXVkZSBGYWJsZSA1LjEgTWF4LiBMYXVuY2ggY29tcGFyaXNvbjsgcGVyLW1vZGVsIGFnZW50IGhhcm5lc3MsIGV2YWx1YXRpb24gYnVkZ2V0IGFuZCByZXJ1biBwcm92ZW5hbmNlIGFyZSBub3Qgc3BlY2lmaWVkLg","name":"Grok 4.7 launch comparison","notes":"Grok 4.7 XHigh (DeepSWE: High); Grok 4.6 High; GPT-5.6 Sol Max; Claude Fable 5.1 Max. Launch comparison; per-model agent harness, evaluation budget and rerun provenance are not specified."},{"benchmarkId":"healthbench-professional","id":"healthbench-professional:xai-grok47:R3JvayA0LjcgWEhpZ2ggKERlZXBTV0U6IEhpZ2gpOyBHcm9rIDQuNiBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IENsYXVkZSBGYWJsZSA1LjEgTWF4LiBMYXVuY2ggY29tcGFyaXNvbjsgcGVyLW1vZGVsIGFnZW50IGhhcm5lc3MsIGV2YWx1YXRpb24gYnVkZ2V0IGFuZCByZXJ1biBwcm92ZW5hbmNlIGFyZSBub3Qgc3BlY2lmaWVkLg","name":"Grok 4.7 launch comparison","notes":"Grok 4.7 XHigh (DeepSWE: High); Grok 4.6 High; GPT-5.6 Sol Max; Claude Fable 5.1 Max. Launch comparison; per-model agent harness, evaluation budget and rerun provenance are not specified."},{"benchmarkId":"gdpval-aa-2","id":"gdpval-aa-2:xai-grok47:R3JvayA0LjcgWEhpZ2g7IEdyb2sgNC42IEhpZ2g7IENsYXVkZSBGYWJsZSA1LjEgTWF4OyBHUFQtNiBBc3RyYSBNYXguIEdEUHZhbCBsYXVuY2ggY2hhcnQu","name":"Grok 4.7 launch comparison","notes":"Grok 4.7 XHigh; Grok 4.6 High; Claude Fable 5.1 Max; GPT-6 Astra Max. GDPval launch chart."},{"benchmarkId":"swe-bench-pro-percent-higher","id":"swe-bench-pro-percent-higher:anthropic-opus-5-5-card:QWRhcHRpdmUgdGhpbmtpbmcgYXQgbWF4IGVmZm9ydDsgZGVmYXVsdCBzYW1wbGluZzsgZml2ZS10cmlhbCBtZWFuOyBjb250ZXh0IGF0IG1vc3QgMU0uIFNXRS1iZW5jaCBQcm8gcHJvYmxlbXMgZnJvbSBhY3RpdmVseSBtYWludGFpbmVkIHJlcG9zaXRvcmllcyB3aXRoIGxhcmdlIG11bHRpLWZpbGUgZGlmZnMu","name":"Claude Opus 5.5 System Card","notes":"Adaptive thinking at max effort; default sampling; five-trial mean; context at most 1M. SWE-bench Pro problems from actively maintained repositories with large multi-file diffs."},{"benchmarkId":"swe-bench-multilingual-percent","id":"swe-bench-multilingual-percent:anthropic-opus-5-5-card:QWRhcHRpdmUgdGhpbmtpbmcgYXQgbWF4IGVmZm9ydDsgZGVmYXVsdCBzYW1wbGluZzsgZml2ZS10cmlhbCBtZWFuOyBjb250ZXh0IGF0IG1vc3QgMU0uIDMwMCBwcm9ibGVtcyBhY3Jvc3MgbmluZSBwcm9ncmFtbWluZyBsYW5ndWFnZXMu","name":"Claude Opus 5.5 System Card","notes":"Adaptive thinking at max effort; default sampling; five-trial mean; context at most 1M. 300 problems across nine programming languages."},{"benchmarkId":"swe-bench-multimodal","id":"swe-bench-multimodal:anthropic-opus-5-5-card:QWRhcHRpdmUgdGhpbmtpbmcgYXQgbWF4IGVmZm9ydDsgZGVmYXVsdCBzYW1wbGluZzsgZml2ZS10cmlhbCBtZWFuOyBjb250ZXh0IGF0IG1vc3QgMU0uIFZpc3VhbCBjb250ZXh0IGFkZGVkIHRvIGlzc3VlIGRlc2NyaXB0aW9ucy4","name":"Claude Opus 5.5 System Card","notes":"Adaptive thinking at max effort; default sampling; five-trial mean; context at most 1M. Visual context added to issue descriptions."},{"benchmarkId":"deepswe-1-1-percent","id":"deepswe-1-1-percent:anthropic-opus-5-5-card:MTEzIGxvbmctaG9yaXpvbiB0YXNrczsgZml2ZS10cmlhbCBtZWFuLiBTZWN0aW9uIDguMyBkb2VzIG5vdCBzdGF0ZSByZWFzb25pbmcgZWZmb3J0Lg","name":"Claude Opus 5.5 System Card","notes":"113 long-horizon tasks; five-trial mean. Section 8.3 does not state reasoning effort."},{"benchmarkId":"frontiercode-1-1-main","id":"frontiercode-1-1-main:anthropic-opus-5-5-card:Q29nbml0aW9uIGFnZW50aWMgY29kaW5nIGluIENsYXVkZSBDb2RlOyBjb21wb3NpdGUgZnVuY3Rpb25hbCBhbmQgY29kZS1xdWFsaXR5IHNjb3JlOyBtYXggZWZmb3J0OyBtZWFuQDUuIENvZ25pdGlvbiByYW4gdGhlIGV2YWx1YXRpb24u","name":"Claude Opus 5.5 System Card","notes":"Cognition agentic coding in Claude Code; composite functional and code-quality score; max effort; mean@5. Cognition ran the evaluation."},{"benchmarkId":"frontiercode-1-1-main","id":"frontiercode-1-1-main:anthropic-opus-5-5-card:Q29nbml0aW9uIGFnZW50aWMgY29kaW5nIGluIENsYXVkZSBDb2RlOyBjb21wb3NpdGUgZnVuY3Rpb25hbCBhbmQgY29kZS1xdWFsaXR5IHNjb3JlOyBtZWRpdW0gZWZmb3J0OyBtZWFuQDUuIEhpZ2hlc3QgTWFpbiBzY29yZS4gQ29nbml0aW9uIHJhbiB0aGUgZXZhbHVhdGlvbi4","name":"Claude Opus 5.5 System Card","notes":"Cognition agentic coding in Claude Code; composite functional and code-quality score; medium effort; mean@5. Highest Main score. Cognition ran the evaluation."},{"benchmarkId":"frontiercode-1-1-extended","id":"frontiercode-1-1-extended:anthropic-opus-5-5-card:Q29nbml0aW9uIGFnZW50aWMgY29kaW5nIGluIENsYXVkZSBDb2RlOyBjb21wb3NpdGUgZnVuY3Rpb25hbCBhbmQgY29kZS1xdWFsaXR5IHNjb3JlOyBtZWRpdW0gZWZmb3J0OyBtZWFuQDUuIEhpZ2hlc3QgRXh0ZW5kZWQgc2NvcmUuIENvZ25pdGlvbiByYW4gdGhlIGV2YWx1YXRpb24u","name":"Claude Opus 5.5 System Card","notes":"Cognition agentic coding in Claude Code; composite functional and code-quality score; medium effort; mean@5. Highest Extended score. Cognition ran the evaluation."},{"benchmarkId":"frontiercode-1-1-extended","id":"frontiercode-1-1-extended:anthropic-opus-5-5-card:Q29nbml0aW9uIGFnZW50aWMgY29kaW5nIGluIENsYXVkZSBDb2RlOyBjb21wb3NpdGUgZnVuY3Rpb25hbCBhbmQgY29kZS1xdWFsaXR5IHNjb3JlOyBtYXggZWZmb3J0OyBtZWFuQDUuIENvZ25pdGlvbiByYW4gdGhlIGV2YWx1YXRpb24u","name":"Claude Opus 5.5 System Card","notes":"Cognition agentic coding in Claude Code; composite functional and code-quality score; max effort; mean@5. Cognition ran the evaluation."},{"benchmarkId":"terminal-bench-4-0-percent","id":"terminal-bench-4-0-percent:anthropic-opus-5-5-card:Q2xhdWRlIENvZGUgLS1iYXJlOyB4aGlnaCB0aGlua2luZyBlZmZvcnQ7IHNhZmVndWFyZHMgZW5hYmxlZCB3aXRoIHNlcnZlci1zaWRlIGZhbGxiYWNrICgyLjUlIG9mIHJlcXVlc3RzLCAxMCUgb2YgdHJpYWxzKTsgZml2ZSB0cmlhbHMgcGVyIHRhc2sgKDMzMCB0cmlhbHMpIG9uIDY2IHRhc2tzLg","name":"Claude Opus 5.5 System Card","notes":"Claude Code --bare; xhigh thinking effort; safeguards enabled with server-side fallback (2.5% of requests, 10% of trials); five trials per task (330 trials) on 66 tasks."},{"benchmarkId":"terminal-bench-4-0-percent","id":"terminal-bench-4-0-percent:anthropic-opus-5-5-card:Q2xhdWRlIENvZGUgLS1iYXJlOyBtYXggdGhpbmtpbmcgZWZmb3J0OyBzYWZlZ3VhcmRzIGVuYWJsZWQgd2l0aCB0aGUgZGVmYXVsdCBzZXJ2ZXItc2lkZSBmYWxsYmFjazsgZml2ZSB0cmlhbHMgcGVyIHRhc2sgb24gNjYgdGFza3MuIFNlY3Rpb24gOC41IHNheXMgdGhpcyBpcyB3aXRoaW4gbm9pc2Ugb2YgeGhpZ2gu","name":"Claude Opus 5.5 System Card","notes":"Claude Code --bare; max thinking effort; safeguards enabled with the default server-side fallback; five trials per task on 66 tasks. Section 8.5 says this is within noise of xhigh."},{"benchmarkId":"terminal-bench-science-0-1-percent","id":"terminal-bench-science-0-1-percent:anthropic-opus-5-5-card:Q2xhdWRlIENvZGUgLS1iYXJlOyBtYXggdGhpbmtpbmcgZWZmb3J0OyBzYWZlZ3VhcmRzIGVuYWJsZWQgd2l0aCBzZXJ2ZXItc2lkZSBmYWxsYmFjayAoMy45JSBvZiByZXF1ZXN0cywgNSUgb2YgdHJpYWxzKTsgMTAgdHJpYWxzIHBlciB0YXNrICg3MDAgdHJpYWxzKSBvbiA3MCB0YXNrcy4","name":"Claude Opus 5.5 System Card","notes":"Claude Code --bare; max thinking effort; safeguards enabled with server-side fallback (3.9% of requests, 5% of trials); 10 trials per task (700 trials) on 70 tasks."},{"benchmarkId":"frontierswe-2-percent","id":"frontierswe-2-percent:anthropic-opus-5-5-card:UHJveGltYWwgYWdlbnQgaGFybmVzczsgbWF4IHJlYXNvbmluZyBlZmZvcnQ7IDM0IHRhc2tzOyBmaXZlIHRyaWFscyBwZXIgdGFzazsgbWVhbiBhY3Jvc3MgdHJpYWxzLg","name":"Claude Opus 5.5 System Card","notes":"Proximal agent harness; max reasoning effort; 34 tasks; five trials per task; mean across trials."},{"benchmarkId":"cursorbench-4-0-percent","id":"cursorbench-4-0-percent:anthropic-opus-5-5-card:Q3Vyc29yIHByb2R1Y3Rpb24gYWdlbnQgaGFybmVzczsgbWF4IGVmZm9ydC4gSW5kZXBlbmRlbnRseSBtZWFzdXJlZCBieSBDdXJzb3I7IEFudGhyb3BpYyBlc3RpbWF0ZWQgY29zdCBmcm9tIEN1cnNvciB0b2tlbiBjb3VudHMu","name":"Claude Opus 5.5 System Card","notes":"Cursor production agent harness; max effort. Independently measured by Cursor; Anthropic estimated cost from Cursor token counts."},{"benchmarkId":"cursorbench-4-0-percent","id":"cursorbench-4-0-percent:anthropic-opus-5-5-card:Q3Vyc29yIHByb2R1Y3Rpb24gYWdlbnQgaGFybmVzczsgeGhpZ2ggZWZmb3J0LiBJbmRlcGVuZGVudGx5IG1lYXN1cmVkIGJ5IEN1cnNvci4","name":"Claude Opus 5.5 System Card","notes":"Cursor production agent harness; xhigh effort. Independently measured by Cursor."},{"benchmarkId":"cursorbench-4-0-percent","id":"cursorbench-4-0-percent:anthropic-opus-5-5-card:Q3Vyc29yIHByb2R1Y3Rpb24gYWdlbnQgaGFybmVzczsgaGlnaCBlZmZvcnQuIEluZGVwZW5kZW50bHkgbWVhc3VyZWQgYnkgQ3Vyc29yLg","name":"Claude Opus 5.5 System Card","notes":"Cursor production agent harness; high effort. Independently measured by Cursor."},{"benchmarkId":"cursorbench-4-0-percent","id":"cursorbench-4-0-percent:anthropic-opus-5-5-card:Q3Vyc29yIHByb2R1Y3Rpb24gYWdlbnQgaGFybmVzczsgbWVkaXVtIGVmZm9ydC4gSW5kZXBlbmRlbnRseSBtZWFzdXJlZCBieSBDdXJzb3Iu","name":"Claude Opus 5.5 System Card","notes":"Cursor production agent harness; medium effort. Independently measured by Cursor."},{"benchmarkId":"programbench-166-golden-task-subset","id":"programbench-166-golden-task-subset:anthropic-opus-5-5-card:bWluaS1zd2UtYWdlbnQgd2l0aG91dCB0aGUgc2l4LWhvdXIgdGltZW91dDsgMzQgdGFza3Mgd2l0aCBhIHJlZmVyZW5jZSBiaW5hcnkgYmVsb3cgMC45IGV4Y2x1ZGVkOyBzY29yZWQgb25seSBvbiB0ZXN0cyB0aGUgcmVmZXJlbmNlIGJpbmFyeSBwYXNzZXM7IGNvbnRleHQgdXAgdG8gMU0uIFNlY3Rpb24gOC4xMC4xIGRvZXMgbm90IHN0YXRlIHJlYXNvbmluZyBlZmZvcnQu","name":"Claude Opus 5.5 System Card","notes":"mini-swe-agent without the six-hour timeout; 34 tasks with a reference binary below 0.9 excluded; scored only on tests the reference binary passes; context up to 1M. Section 8.10.1 does not state reasoning effort."},{"benchmarkId":"humanity-s-last-exam","id":"humanity-s-last-exam:anthropic-opus-5-5-card:RnVsbCAyLDUwMCBxdWVzdGlvbnM7IG5vIHRvb2xzOyBzZWN0aW9uIDguMTEuMSBzZXRzIHRoaW5raW5nIHRvIGF1dG8sIGEgMU0gdG90YWwgdG9rZW4gY2FwLCBubyBjb21wYWN0aW9uLCBhbmQgYW4gT3B1cyA0LjYgZ3JhZGVyLg","name":"Claude Opus 5.5 System Card","notes":"Full 2,500 questions; no tools; section 8.11.1 sets thinking to auto, a 1M total token cap, no compaction, and an Opus 4.6 grader."},{"benchmarkId":"humanity-s-last-exam","id":"humanity-s-last-exam:anthropic-opus-5-5-card:RnVsbCAyLDUwMCBxdWVzdGlvbnM7IHdlYiBzZWFyY2gsIHdlYiBmZXRjaCwgcHJvZ3JhbW1hdGljIHRvb2wgY2FsbGluZywgYW5kIGNvZGUgZXhlY3V0aW9uOyB0aGlua2luZyBzZXQgdG8gYXV0bzsgMU0gdG90YWwgdG9rZW4gY2FwOyBubyBjb21wYWN0aW9uOyBPcHVzIDQuNiBncmFkZXI7IEhMRSBzb3VyY2UgYmxvY2tsaXN0IGFuZCBjb250YW1pbmF0aW9uIHJldmlldy4","name":"Claude Opus 5.5 System Card","notes":"Full 2,500 questions; web search, web fetch, programmatic tool calling, and code execution; thinking set to auto; 1M total token cap; no compaction; Opus 4.6 grader; HLE source blocklist and contamination review."},{"benchmarkId":"chartography","id":"chartography:anthropic-opus-5-5-card:MTAwIHRhc2tzOyBhZGFwdGl2ZSB0aGlua2luZyBhdCBtYXggZWZmb3J0OyBmaXZlIHJ1bnM7IG5vIHRvb2xzOyBHZW1pbmkgMy41IEZsYXNoIGdyYWRlcjsgZXhwZXJ0IGFjY2VwdGFibGUgcmFuZ2VzLg","name":"Claude Opus 5.5 System Card","notes":"100 tasks; adaptive thinking at max effort; five runs; no tools; Gemini 3.5 Flash grader; expert acceptable ranges."},{"benchmarkId":"chartography","id":"chartography:anthropic-opus-5-5-card:MTAwIHRhc2tzOyBhZGFwdGl2ZSB0aGlua2luZyBhdCBtYXggZWZmb3J0OyBmaXZlIHJ1bnM7IHdpdGggdG9vbHMgKGNvbnRhaW5lciwgaW1hZ2UgZmlsZSwgc3RhbmRhcmQgbGlicmFyaWVzLCBhbmQgYW4gaW1hZ2UgY3JvcHBpbmcgdG9vbCk7IEdlbWluaSAzLjUgRmxhc2ggZ3JhZGVyLg","name":"Claude Opus 5.5 System Card","notes":"100 tasks; adaptive thinking at max effort; five runs; with tools (container, image file, standard libraries, and an image cropping tool); Gemini 3.5 Flash grader."},{"benchmarkId":"benchcad-vision2code-1000-file-subset","id":"benchcad-vision2code-1000-file-subset:anthropic-opus-5-5-card:UmFuZG9tIDEsMDAwIG9mIDE3LDkwMCBWaXNpb24yQ29kZSBmaWxlczsgZml2ZSBydW5zOyBhZGFwdGl2ZSB0aGlua2luZyBhdCBtYXggZWZmb3J0OyBubyB0b29sczsgdmlld3MgcmVuZGVyZWQgYXQgMjU2eDI1NiBweCwgdGhlIHJlc29sdXRpb24gQW50aHJvcGljIHNheXMgbWF0Y2hlcyB0aGUgcmVmZXJlbmNlIGltcGxlbWVudGF0aW9uLg","name":"Claude Opus 5.5 System Card","notes":"Random 1,000 of 17,900 Vision2Code files; five runs; adaptive thinking at max effort; no tools; views rendered at 256x256 px, the resolution Anthropic says matches the reference implementation."},{"benchmarkId":"benchcad-vision2code-1000-file-subset","id":"benchcad-vision2code-1000-file-subset:anthropic-opus-5-5-card:UmFuZG9tIDEsMDAwIG9mIDE3LDkwMCBWaXNpb24yQ29kZSBmaWxlczsgZml2ZSBydW5zOyBhZGFwdGl2ZSB0aGlua2luZyBhdCBtYXggZWZmb3J0OyB3aXRoIHRvb2xzIChjb250YWluZXIsIGltYWdlIGZpbGVzLCBzdGFuZGFyZCBsaWJyYXJpZXMsIGFuZCBhbiBpbWFnZSBjcm9wcGluZyB0b29sKTsgMjU2eDI1NiBweCB2aWV3cy4","name":"Claude Opus 5.5 System Card","notes":"Random 1,000 of 17,900 Vision2Code files; five runs; adaptive thinking at max effort; with tools (container, image files, standard libraries, and an image cropping tool); 256x256 px views."},{"benchmarkId":"osworld-2-0-september-10-2026-task-release","id":"osworld-2-0-september-10-2026-task-release:anthropic-opus-5-5-card:UGFydGlhbCBzY29yZSwgcGFzc0AxOyAxMDggdGFza3M7IGZpdmUgcnVuczsgMTA4MHA7IDUwMCBhY3Rpb24gc3RlcHM7IG1heCByZWFzb25pbmcgZWZmb3J0OyBPcHVzIDQuOCBncmFkZXIgd2hlcmUgcmVxdWlyZWQuIFNlcHRlbWJlciAxMCwgMjAyNiB0YXNrIGZpbGVzIGFuZCBzZXJ2ZXItc2lkZSBjb250ZXh0IGNvbXBhY3Rpb24gYWZ0ZXIgMTAwayB0b2tlbnMu","name":"Claude Opus 5.5 System Card","notes":"Partial score, pass@1; 108 tasks; five runs; 1080p; 500 action steps; max reasoning effort; Opus 4.8 grader where required. September 10, 2026 task files and server-side context compaction after 100k tokens."},{"benchmarkId":"osworld-2-0-september-10-2026-task-release","id":"osworld-2-0-september-10-2026-task-release:anthropic-opus-5-5-card:U3RyaWN0IHBhc3MgcmF0ZSwgcGFzc0AxOyAxMDggdGFza3M7IGZpdmUgcnVuczsgMTA4MHA7IDUwMCBhY3Rpb24gc3RlcHM7IG1heCByZWFzb25pbmcgZWZmb3J0OyBPcHVzIDQuOCBncmFkZXIgd2hlcmUgcmVxdWlyZWQuIFNlcHRlbWJlciAxMCwgMjAyNiB0YXNrIGZpbGVzIGFuZCBzZXJ2ZXItc2lkZSBjb250ZXh0IGNvbXBhY3Rpb24gYWZ0ZXIgMTAwayB0b2tlbnMu","name":"Claude Opus 5.5 System Card","notes":"Strict pass rate, pass@1; 108 tasks; five runs; 1080p; 500 action steps; max reasoning effort; Opus 4.8 grader where required. September 10, 2026 task files and server-side context compaction after 100k tokens."},{"benchmarkId":"officeqa","id":"officeqa:anthropic-opus-5-5-card:QWdlbnRpYyBleHRyYWN0ZWQtdGV4dCBUcmVhc3VyeSBCdWxsZXRpbiBjb3JwdXMgd2l0aCBjb2RlIGV4ZWN1dGlvbjsgbWF4IGVmZm9ydDsgbWVhbiBvZiBmaXZlIHJ1bnMu","name":"Claude Opus 5.5 System Card","notes":"Agentic extracted-text Treasury Bulletin corpus with code execution; max effort; mean of five runs."},{"benchmarkId":"officeqa-pro","id":"officeqa-pro:anthropic-opus-5-5-card:SGFyZGVyIDEzMy1xdWVzdGlvbiBzdWJzZXQ7IGFnZW50aWMgZXh0cmFjdGVkLXRleHQgVHJlYXN1cnkgQnVsbGV0aW4gY29ycHVzIHdpdGggY29kZSBleGVjdXRpb247IG1heCBlZmZvcnQ7IG1lYW4gb2YgZml2ZSBydW5zLg","name":"Claude Opus 5.5 System Card","notes":"Harder 133-question subset; agentic extracted-text Treasury Bulletin corpus with code execution; max effort; mean of five runs."},{"benchmarkId":"legal-agent-benchmark-120-task-held-out-subset","id":"legal-agent-benchmark-120-task-held-out-subset:anthropic-opus-5-5-card:QWxsLXBhc3MgcmF0ZSBvbiBIYXJ2ZXkncyBoZWxkLW91dCAxMjAgdGFza3M7IG1heCBlZmZvcnQuIEFydGlmaWNpYWwgQW5hbHlzaXMgaGFybmVzcywgYXMgaW4gdGhlIHByZXZpb3VzIHN5c3RlbSBjYXJkLg","name":"Claude Opus 5.5 System Card","notes":"All-pass rate on Harvey's held-out 120 tasks; max effort. Artificial Analysis harness, as in the previous system card."},{"benchmarkId":"legal-agent-benchmark-120-task-held-out-subset","id":"legal-agent-benchmark-120-task-held-out-subset:anthropic-opus-5-5-card:TWVhbiBjcml0ZXJpb24tcGFzcyByYXRlIG9uIEhhcnZleSdzIGhlbGQtb3V0IDEyMCB0YXNrczsgbWF4IGVmZm9ydC4gQXJ0aWZpY2lhbCBBbmFseXNpcyBoYXJuZXNzLCBhcyBpbiB0aGUgcHJldmlvdXMgc3lzdGVtIGNhcmQu","name":"Claude Opus 5.5 System Card","notes":"Mean criterion-pass rate on Harvey's held-out 120 tasks; max effort. Artificial Analysis harness, as in the previous system card."},{"benchmarkId":"gdpval-aa-2-1","id":"gdpval-aa-2-1:anthropic-opus-5-5-card:QXJ0aWZpY2lhbCBBbmFseXNpcyBHRFB2YWwtQUEgdjIuMTsgMjIwIEdEUHZhbCBnb2xkIHRhc2tzOyBibGluZCBwYWlyd2lzZSBFbG8gYW5jaG9yZWQgdG8gRGVlcFNlZWsgVjQuMSBGbGFzaCAobWF4KSBhdCAxNjAwOyBtYXggZWZmb3J0LiBSdW4gYnkgQXJ0aWZpY2lhbCBBbmFseXNpcy4","name":"Claude Opus 5.5 System Card","notes":"Artificial Analysis GDPval-AA v2.1; 220 GDPval gold tasks; blind pairwise Elo anchored to DeepSeek V4.1 Flash (max) at 1600; max effort. Run by Artificial Analysis."},{"benchmarkId":"gdpval-aa-2-1","id":"gdpval-aa-2-1:anthropic-opus-5-5-card:QXJ0aWZpY2lhbCBBbmFseXNpcyBHRFB2YWwtQUEgdjIuMTsgMjIwIEdEUHZhbCBnb2xkIHRhc2tzOyBibGluZCBwYWlyd2lzZSBFbG8gYW5jaG9yZWQgdG8gRGVlcFNlZWsgVjQuMSBGbGFzaCAobWF4KSBhdCAxNjAwOyB4aGlnaCBlZmZvcnQuIFJ1biBieSBBcnRpZmljaWFsIEFuYWx5c2lzLg","name":"Claude Opus 5.5 System Card","notes":"Artificial Analysis GDPval-AA v2.1; 220 GDPval gold tasks; blind pairwise Elo anchored to DeepSeek V4.1 Flash (max) at 1600; xhigh effort. Run by Artificial Analysis."},{"benchmarkId":"aa-briefcase-1-1-elo","id":"aa-briefcase-1-1-elo:anthropic-opus-5-5-card:QXJ0aWZpY2lhbCBBbmFseXNpcyBBQS1CcmllZmNhc2UgdjEuMTsgbG9uZy1ob3Jpem9uIGtub3dsZWRnZSBwcm9qZWN0czsgcnVicmljIHNjb3JpbmcgYW5kIHBhaXJ3aXNlIGp1ZGdpbmc7IG1heCBlZmZvcnQuIFJ1biBieSBBcnRpZmljaWFsIEFuYWx5c2lzLg","name":"Claude Opus 5.5 System Card","notes":"Artificial Analysis AA-Briefcase v1.1; long-horizon knowledge projects; rubric scoring and pairwise judging; max effort. Run by Artificial Analysis."},{"benchmarkId":"aa-briefcase-1-1-elo","id":"aa-briefcase-1-1-elo:anthropic-opus-5-5-card:QXJ0aWZpY2lhbCBBbmFseXNpcyBBQS1CcmllZmNhc2UgdjEuMTsgbG9uZy1ob3Jpem9uIGtub3dsZWRnZSBwcm9qZWN0czsgcnVicmljIHNjb3JpbmcgYW5kIHBhaXJ3aXNlIGp1ZGdpbmc7IHhoaWdoIGVmZm9ydC4gUnVuIGJ5IEFydGlmaWNpYWwgQW5hbHlzaXMu","name":"Claude Opus 5.5 System Card","notes":"Artificial Analysis AA-Briefcase v1.1; long-horizon knowledge projects; rubric scoring and pairwise judging; xhigh effort. Run by Artificial Analysis."},{"benchmarkId":"aa-briefcase-1-1-elo","id":"aa-briefcase-1-1-elo:anthropic-opus-5-5-card:QXJ0aWZpY2lhbCBBbmFseXNpcyBBQS1CcmllZmNhc2UgdjEuMTsgbG9uZy1ob3Jpem9uIGtub3dsZWRnZSBwcm9qZWN0czsgcnVicmljIHNjb3JpbmcgYW5kIHBhaXJ3aXNlIGp1ZGdpbmc7IGhpZ2ggZWZmb3J0LiBSdW4gYnkgQXJ0aWZpY2lhbCBBbmFseXNpcy4","name":"Claude Opus 5.5 System Card","notes":"Artificial Analysis AA-Briefcase v1.1; long-horizon knowledge projects; rubric scoring and pairwise judging; high effort. Run by Artificial Analysis."},{"benchmarkId":"toolathlon-verified-june2026","id":"toolathlon-verified-june2026:anthropic-opus-5-5-card:UGFzc0AxOyAxMDggdGFza3M7IHRocmVlIHRyaWFsczsgaW50ZXJuYWwgaGFybmVzcyBtaXJyb3JpbmcgVG9vbGF0aGxvbi1WZXJpZmllZDsgYWRhcHRpdmUgdGhpbmtpbmcgYXQgbWF4IGVmZm9ydDsgc2FmZXR5IGNsYXNzaWZpZXJzIG9uOyBvbmUgc2FmZXR5IHN0b3AgYW5kIHNpeCBzYW5kYm94LW1vbml0b3IgaGFsdHMgY291bnRlZCBhcyBmYWlsdXJlcy4","name":"Claude Opus 5.5 System Card","notes":"Pass@1; 108 tasks; three trials; internal harness mirroring Toolathlon-Verified; adaptive thinking at max effort; safety classifiers on; one safety stop and six sandbox-monitor halts counted as failures."},{"benchmarkId":"toolathlon-verified-june2026","id":"toolathlon-verified-june2026:anthropic-opus-5-5-card:UGFzc0AzIChhdCBsZWFzdCBvbmUgb2YgdGhyZWUgdHJpYWxzIGNvcnJlY3QpOyAxMDggdGFza3M7IGludGVybmFsIGhhcm5lc3M7IGFkYXB0aXZlIHRoaW5raW5nIGF0IG1heCBlZmZvcnQ7IHNhZmV0eSBjbGFzc2lmaWVycyBvbi4","name":"Claude Opus 5.5 System Card","notes":"Pass@3 (at least one of three trials correct); 108 tasks; internal harness; adaptive thinking at max effort; safety classifiers on."},{"benchmarkId":"toolathlon-verified-june2026","id":"toolathlon-verified-june2026:anthropic-opus-5-5-card:UGFzc8KzIChhbGwgdGhyZWUgdHJpYWxzIGNvcnJlY3QpOyAxMDggdGFza3M7IGludGVybmFsIGhhcm5lc3M7IGFkYXB0aXZlIHRoaW5raW5nIGF0IG1heCBlZmZvcnQ7IHNhZmV0eSBjbGFzc2lmaWVycyBvbi4","name":"Claude Opus 5.5 System Card","notes":"Pass³ (all three trials correct); 108 tasks; internal harness; adaptive thinking at max effort; safety classifiers on."},{"benchmarkId":"automationbench","id":"automationbench:anthropic-opus-5-5-card:WmFwaWVyIHByaXZhdGUgaGVsZC1vdXQgbGVhZGVyYm9hcmQ7IHNpbXVsYXRlZCBidXNpbmVzcyB3b3JrZmxvd3M7IGV2ZXJ5IGRldGVybWluaXN0aWMgYXNzZXJ0aW9uIG11c3QgcGFzczsgbWF4IGVmZm9ydC4","name":"Claude Opus 5.5 System Card","notes":"Zapier private held-out leaderboard; simulated business workflows; every deterministic assertion must pass; max effort."},{"benchmarkId":"healthbench","id":"healthbench:anthropic-opus-5-5-card:UmF3IHJ1YnJpYyBzY29yZTsgYWRhcHRpdmUgdGhpbmtpbmcgYXQgbWF4IGVmZm9ydDsgZml2ZSB0cmlhbHM7IG5vIHRvb2xzIG9yIGN1c3RvbSBzeXN0ZW0gcHJvbXB0OyBPcHVzIDQuOCBncmFkZXI7IHNhZmV0eSBjbGFzc2lmaWVycyB3aXRoIHJlZnVzYWwgZmFsbGJhY2sgdG8gT3B1cyA1Lg","name":"Claude Opus 5.5 System Card","notes":"Raw rubric score; adaptive thinking at max effort; five trials; no tools or custom system prompt; Opus 4.8 grader; safety classifiers with refusal fallback to Opus 5."},{"benchmarkId":"healthbench","id":"healthbench:anthropic-opus-5-5-card:TGVuZ3RoLWFkanVzdGVkIHNjb3JlIHVzaW5nIHRoZSBHUFQtNS41IHN5c3RlbS1jYXJkIG1ldGhvZDsgb3RoZXJ3aXNlIHRoZSByYXcgSGVhbHRoQmVuY2ggY29uZmlndXJhdGlvbjogYWRhcHRpdmUgbWF4LCBmaXZlIHRyaWFscywgbm8gdG9vbHMsIE9wdXMgNC44IGdyYWRlciwgc2FmZXR5IGZhbGxiYWNrIHRvIE9wdXMgNS4","name":"Claude Opus 5.5 System Card","notes":"Length-adjusted score using the GPT-5.5 system-card method; otherwise the raw HealthBench configuration: adaptive max, five trials, no tools, Opus 4.8 grader, safety fallback to Opus 5."},{"benchmarkId":"healthbench-professional-percent","id":"healthbench-professional-percent:anthropic-opus-5-5-card:UmF3IHJ1YnJpYyBzY29yZTsgYWRhcHRpdmUgdGhpbmtpbmcgYXQgbWF4IGVmZm9ydDsgZml2ZSB0cmlhbHM7IG5vIHRvb2xzIG9yIGN1c3RvbSBzeXN0ZW0gcHJvbXB0OyBPcHVzIDQuOCBncmFkZXI7IHNhZmV0eSBjbGFzc2lmaWVycyB3aXRoIHJlZnVzYWwgZmFsbGJhY2sgdG8gT3B1cyA1Lg","name":"Claude Opus 5.5 System Card","notes":"Raw rubric score; adaptive thinking at max effort; five trials; no tools or custom system prompt; Opus 4.8 grader; safety classifiers with refusal fallback to Opus 5."},{"benchmarkId":"healthbench-professional-percent","id":"healthbench-professional-percent:anthropic-opus-5-5-card:TGVuZ3RoLWFkanVzdGVkIHNjb3JlIHVzaW5nIHRoZSBIZWFsdGhCZW5jaCBQcm9mZXNzaW9uYWwgcGFwZXIgbWV0aG9kOyBhZGFwdGl2ZSBtYXg7IGZpdmUgdHJpYWxzOyBubyB0b29sczsgT3B1cyA0LjggZ3JhZGVyOyBzYWZldHkgZmFsbGJhY2sgdG8gT3B1cyA1Lg","name":"Claude Opus 5.5 System Card","notes":"Length-adjusted score using the HealthBench Professional paper method; adaptive max; five trials; no tools; Opus 4.8 grader; safety fallback to Opus 5."},{"benchmarkId":"gmmlu","id":"gmmlu:anthropic-opus-5-5-card:QXZlcmFnZSBhY2N1cmFjeSBhY3Jvc3MgNDIgbGFuZ3VhZ2VzOyBhZGFwdGl2ZSB0aGlua2luZyBhdCBtYXggZWZmb3J0OyBzaW5nbGUgdHJpYWw7IG5vIHRvb2xzIG9yIGN1c3RvbSBzeXN0ZW0gcHJvbXB0Lg","name":"Claude Opus 5.5 System Card","notes":"Average accuracy across 42 languages; adaptive thinking at max effort; single trial; no tools or custom system prompt."},{"benchmarkId":"milu","id":"milu:anthropic-opus-5-5-card:QXZlcmFnZSBhY2N1cmFjeSBhY3Jvc3MgMTEgbGFuZ3VhZ2VzOyBhZGFwdGl2ZSB0aGlua2luZyBhdCBtYXggZWZmb3J0OyBmaXZlIHRyaWFsczsgbm8gdG9vbHMgb3IgY3VzdG9tIHN5c3RlbSBwcm9tcHQu","name":"Claude Opus 5.5 System Card","notes":"Average accuracy across 11 languages; adaptive thinking at max effort; five trials; no tools or custom system prompt."},{"benchmarkId":"automationbench-1-0-6-percent","id":"automationbench-1-0-6-percent:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBsb3cuIEF1dG9tYXRpb25CZW5jaCAxLjAuNi4gRW5kLXRvLWVuZCB3b3JrZmxvd3MgYWNyb3NzIDQ3IHRvb2xzIGluIHNhbGVzLCBtYXJrZXRpbmcsIG9wZXJhdGlvbnMsIHN1cHBvcnQsIGZpbmFuY2UsIGFuZCBIUi4gVGhlIHBhZ2Ugc2F5cyB0aGUgQ2xhdWRlIEZhYmxlIDUuMSBjb3N0IHBvaW50IG9taXRzIE9wdXMgNSBmYWxsYmFja3MuIE9wZW5BSS1wdWJsaXNoZWQgY2hhcnQgb24gdGhlIEludHJvZHVjaW5nIEdQVC02IFNvbCBhbmQgTHVuYSBwYWdlLg","name":"Introducing GPT-6 Sol and Luna","notes":"Reported reasoning effort low. AutomationBench 1.0.6. End-to-end workflows across 47 tools in sales, marketing, operations, support, finance, and HR. The page says the Claude Fable 5.1 cost point omits Opus 5 fallbacks. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page."},{"benchmarkId":"automationbench-1-0-6-percent","id":"automationbench-1-0-6-percent:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtZWRpdW0uIEF1dG9tYXRpb25CZW5jaCAxLjAuNi4gRW5kLXRvLWVuZCB3b3JrZmxvd3MgYWNyb3NzIDQ3IHRvb2xzIGluIHNhbGVzLCBtYXJrZXRpbmcsIG9wZXJhdGlvbnMsIHN1cHBvcnQsIGZpbmFuY2UsIGFuZCBIUi4gVGhlIHBhZ2Ugc2F5cyB0aGUgQ2xhdWRlIEZhYmxlIDUuMSBjb3N0IHBvaW50IG9taXRzIE9wdXMgNSBmYWxsYmFja3MuIE9wZW5BSS1wdWJsaXNoZWQgY2hhcnQgb24gdGhlIEludHJvZHVjaW5nIEdQVC02IFNvbCBhbmQgTHVuYSBwYWdlLg","name":"Introducing GPT-6 Sol and Luna","notes":"Reported reasoning effort medium. AutomationBench 1.0.6. End-to-end workflows across 47 tools in sales, marketing, operations, support, finance, and HR. The page says the Claude Fable 5.1 cost point omits Opus 5 fallbacks. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page."},{"benchmarkId":"automationbench-1-0-6-percent","id":"automationbench-1-0-6-percent:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBoaWdoLiBBdXRvbWF0aW9uQmVuY2ggMS4wLjYuIEVuZC10by1lbmQgd29ya2Zsb3dzIGFjcm9zcyA0NyB0b29scyBpbiBzYWxlcywgbWFya2V0aW5nLCBvcGVyYXRpb25zLCBzdXBwb3J0LCBmaW5hbmNlLCBhbmQgSFIuIFRoZSBwYWdlIHNheXMgdGhlIENsYXVkZSBGYWJsZSA1LjEgY29zdCBwb2ludCBvbWl0cyBPcHVzIDUgZmFsbGJhY2tzLiBPcGVuQUktcHVibGlzaGVkIGNoYXJ0IG9uIHRoZSBJbnRyb2R1Y2luZyBHUFQtNiBTb2wgYW5kIEx1bmEgcGFnZS4","name":"Introducing GPT-6 Sol and Luna","notes":"Reported reasoning effort high. AutomationBench 1.0.6. End-to-end workflows across 47 tools in sales, marketing, operations, support, finance, and HR. The page says the Claude Fable 5.1 cost point omits Opus 5 fallbacks. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page."},{"benchmarkId":"automationbench-1-0-6-percent","id":"automationbench-1-0-6-percent:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCB4aGlnaC4gQXV0b21hdGlvbkJlbmNoIDEuMC42LiBFbmQtdG8tZW5kIHdvcmtmbG93cyBhY3Jvc3MgNDcgdG9vbHMgaW4gc2FsZXMsIG1hcmtldGluZywgb3BlcmF0aW9ucywgc3VwcG9ydCwgZmluYW5jZSwgYW5kIEhSLiBUaGUgcGFnZSBzYXlzIHRoZSBDbGF1ZGUgRmFibGUgNS4xIGNvc3QgcG9pbnQgb21pdHMgT3B1cyA1IGZhbGxiYWNrcy4gT3BlbkFJLXB1Ymxpc2hlZCBjaGFydCBvbiB0aGUgSW50cm9kdWNpbmcgR1BULTYgU29sIGFuZCBMdW5hIHBhZ2Uu","name":"Introducing GPT-6 Sol and Luna","notes":"Reported reasoning effort xhigh. AutomationBench 1.0.6. End-to-end workflows across 47 tools in sales, marketing, operations, support, finance, and HR. The page says the Claude Fable 5.1 cost point omits Opus 5 fallbacks. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page."},{"benchmarkId":"automationbench-1-0-6-percent","id":"automationbench-1-0-6-percent:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIEF1dG9tYXRpb25CZW5jaCAxLjAuNi4gRW5kLXRvLWVuZCB3b3JrZmxvd3MgYWNyb3NzIDQ3IHRvb2xzIGluIHNhbGVzLCBtYXJrZXRpbmcsIG9wZXJhdGlvbnMsIHN1cHBvcnQsIGZpbmFuY2UsIGFuZCBIUi4gVGhlIHBhZ2Ugc2F5cyB0aGUgQ2xhdWRlIEZhYmxlIDUuMSBjb3N0IHBvaW50IG9taXRzIE9wdXMgNSBmYWxsYmFja3MuIE9wZW5BSS1wdWJsaXNoZWQgY2hhcnQgb24gdGhlIEludHJvZHVjaW5nIEdQVC02IFNvbCBhbmQgTHVuYSBwYWdlLg","name":"Introducing GPT-6 Sol and Luna","notes":"Reported reasoning effort max. AutomationBench 1.0.6. End-to-end workflows across 47 tools in sales, marketing, operations, support, finance, and HR. The page says the Claude Fable 5.1 cost point omits Opus 5 fallbacks. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page."},{"benchmarkId":"agents-last-exam-1","id":"agents-last-exam-1:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBsb3cuIEFnZW50cycgTGFzdCBFeGFtIFYxLiBMb25nLWhvcml6b24gcHJvZmVzc2lvbmFsIHRhc2tzIGFjcm9zcyA1NSBzdWItaW5kdXN0cmllcy4gT3BlbkFJLXB1Ymxpc2hlZCBjaGFydCBvbiB0aGUgSW50cm9kdWNpbmcgR1BULTYgU29sIGFuZCBMdW5hIHBhZ2Uu","name":"Introducing GPT-6 Sol and Luna","notes":"Reported reasoning effort low. Agents' Last Exam V1. Long-horizon professional tasks across 55 sub-industries. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page."},{"benchmarkId":"agents-last-exam-1","id":"agents-last-exam-1:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtZWRpdW0uIEFnZW50cycgTGFzdCBFeGFtIFYxLiBMb25nLWhvcml6b24gcHJvZmVzc2lvbmFsIHRhc2tzIGFjcm9zcyA1NSBzdWItaW5kdXN0cmllcy4gT3BlbkFJLXB1Ymxpc2hlZCBjaGFydCBvbiB0aGUgSW50cm9kdWNpbmcgR1BULTYgU29sIGFuZCBMdW5hIHBhZ2Uu","name":"Introducing GPT-6 Sol and Luna","notes":"Reported reasoning effort medium. Agents' Last Exam V1. Long-horizon professional tasks across 55 sub-industries. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page."},{"benchmarkId":"agents-last-exam-1","id":"agents-last-exam-1:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBoaWdoLiBBZ2VudHMnIExhc3QgRXhhbSBWMS4gTG9uZy1ob3Jpem9uIHByb2Zlc3Npb25hbCB0YXNrcyBhY3Jvc3MgNTUgc3ViLWluZHVzdHJpZXMuIE9wZW5BSS1wdWJsaXNoZWQgY2hhcnQgb24gdGhlIEludHJvZHVjaW5nIEdQVC02IFNvbCBhbmQgTHVuYSBwYWdlLg","name":"Introducing GPT-6 Sol and Luna","notes":"Reported reasoning effort high. Agents' Last Exam V1. Long-horizon professional tasks across 55 sub-industries. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page."},{"benchmarkId":"agents-last-exam-1","id":"agents-last-exam-1:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCB4aGlnaC4gQWdlbnRzJyBMYXN0IEV4YW0gVjEuIExvbmctaG9yaXpvbiBwcm9mZXNzaW9uYWwgdGFza3MgYWNyb3NzIDU1IHN1Yi1pbmR1c3RyaWVzLiBPcGVuQUktcHVibGlzaGVkIGNoYXJ0IG9uIHRoZSBJbnRyb2R1Y2luZyBHUFQtNiBTb2wgYW5kIEx1bmEgcGFnZS4","name":"Introducing GPT-6 Sol and Luna","notes":"Reported reasoning effort xhigh. Agents' Last Exam V1. Long-horizon professional tasks across 55 sub-industries. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page."},{"benchmarkId":"agents-last-exam-1","id":"agents-last-exam-1:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIEFnZW50cycgTGFzdCBFeGFtIFYxLiBMb25nLWhvcml6b24gcHJvZmVzc2lvbmFsIHRhc2tzIGFjcm9zcyA1NSBzdWItaW5kdXN0cmllcy4gT3BlbkFJLXB1Ymxpc2hlZCBjaGFydCBvbiB0aGUgSW50cm9kdWNpbmcgR1BULTYgU29sIGFuZCBMdW5hIHBhZ2Uu","name":"Introducing GPT-6 Sol and Luna","notes":"Reported reasoning effort max. Agents' Last Exam V1. Long-horizon professional tasks across 55 sub-industries. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page."},{"benchmarkId":"frontiercode-1-1-main-score-1-1-main","id":"frontiercode-1-1-main-score-1-1-main:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBsb3cuIEZyb250aWVyQ29kZSAxLjEgTWFpbiwgYXMgaWRlbnRpZmllZCBpbiB0aGUgc3Vycm91bmRpbmcgYXJ0aWNsZS4gVGhlIGNoYXJ0IHRpdGxlIGlzIEZyb250aWVyQ29kZS4gR3JhZGVkIG9uIGNvcnJlY3RuZXNzIGFuZCBtZXJnZWFiaWxpdHkuIE9wZW5BSS1wdWJsaXNoZWQgY2hhcnQgb24gdGhlIEludHJvZHVjaW5nIEdQVC02IFNvbCBhbmQgTHVuYSBwYWdlLg","name":"Introducing GPT-6 Sol and Luna","notes":"Reported reasoning effort low. FrontierCode 1.1 Main, as identified in the surrounding article. The chart title is FrontierCode. Graded on correctness and mergeability. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page."},{"benchmarkId":"frontiercode-1-1-main-score-1-1-main","id":"frontiercode-1-1-main-score-1-1-main:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtZWRpdW0uIEZyb250aWVyQ29kZSAxLjEgTWFpbiwgYXMgaWRlbnRpZmllZCBpbiB0aGUgc3Vycm91bmRpbmcgYXJ0aWNsZS4gVGhlIGNoYXJ0IHRpdGxlIGlzIEZyb250aWVyQ29kZS4gR3JhZGVkIG9uIGNvcnJlY3RuZXNzIGFuZCBtZXJnZWFiaWxpdHkuIE9wZW5BSS1wdWJsaXNoZWQgY2hhcnQgb24gdGhlIEludHJvZHVjaW5nIEdQVC02IFNvbCBhbmQgTHVuYSBwYWdlLg","name":"Introducing GPT-6 Sol and Luna","notes":"Reported reasoning effort medium. FrontierCode 1.1 Main, as identified in the surrounding article. The chart title is FrontierCode. Graded on correctness and mergeability. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page."},{"benchmarkId":"frontiercode-1-1-main-score-1-1-main","id":"frontiercode-1-1-main-score-1-1-main:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBoaWdoLiBGcm9udGllckNvZGUgMS4xIE1haW4sIGFzIGlkZW50aWZpZWQgaW4gdGhlIHN1cnJvdW5kaW5nIGFydGljbGUuIFRoZSBjaGFydCB0aXRsZSBpcyBGcm9udGllckNvZGUuIEdyYWRlZCBvbiBjb3JyZWN0bmVzcyBhbmQgbWVyZ2VhYmlsaXR5LiBPcGVuQUktcHVibGlzaGVkIGNoYXJ0IG9uIHRoZSBJbnRyb2R1Y2luZyBHUFQtNiBTb2wgYW5kIEx1bmEgcGFnZS4","name":"Introducing GPT-6 Sol and Luna","notes":"Reported reasoning effort high. FrontierCode 1.1 Main, as identified in the surrounding article. The chart title is FrontierCode. Graded on correctness and mergeability. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page."},{"benchmarkId":"frontiercode-1-1-main-score-1-1-main","id":"frontiercode-1-1-main-score-1-1-main:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCB4aGlnaC4gRnJvbnRpZXJDb2RlIDEuMSBNYWluLCBhcyBpZGVudGlmaWVkIGluIHRoZSBzdXJyb3VuZGluZyBhcnRpY2xlLiBUaGUgY2hhcnQgdGl0bGUgaXMgRnJvbnRpZXJDb2RlLiBHcmFkZWQgb24gY29ycmVjdG5lc3MgYW5kIG1lcmdlYWJpbGl0eS4gT3BlbkFJLXB1Ymxpc2hlZCBjaGFydCBvbiB0aGUgSW50cm9kdWNpbmcgR1BULTYgU29sIGFuZCBMdW5hIHBhZ2Uu","name":"Introducing GPT-6 Sol and Luna","notes":"Reported reasoning effort xhigh. FrontierCode 1.1 Main, as identified in the surrounding article. The chart title is FrontierCode. Graded on correctness and mergeability. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page."},{"benchmarkId":"frontiercode-1-1-main-score-1-1-main","id":"frontiercode-1-1-main-score-1-1-main:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIEZyb250aWVyQ29kZSAxLjEgTWFpbiwgYXMgaWRlbnRpZmllZCBpbiB0aGUgc3Vycm91bmRpbmcgYXJ0aWNsZS4gVGhlIGNoYXJ0IHRpdGxlIGlzIEZyb250aWVyQ29kZS4gR3JhZGVkIG9uIGNvcnJlY3RuZXNzIGFuZCBtZXJnZWFiaWxpdHkuIE9wZW5BSS1wdWJsaXNoZWQgY2hhcnQgb24gdGhlIEludHJvZHVjaW5nIEdQVC02IFNvbCBhbmQgTHVuYSBwYWdlLg","name":"Introducing GPT-6 Sol and Luna","notes":"Reported reasoning effort max. FrontierCode 1.1 Main, as identified in the surrounding article. The chart title is FrontierCode. Graded on correctness and mergeability. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page."},{"benchmarkId":"deepswe-v1-1-1-1-percent","id":"deepswe-v1-1-1-1-percent:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBsb3cuIERlZXBTV0UgdjEuMS4gT3JpZ2luYWwgbG9uZy1ob3Jpem9uIHNvZnR3YXJlLWVuZ2luZWVyaW5nIHRhc2tzLiBPcGVuQUktcHVibGlzaGVkIGNoYXJ0IG9uIHRoZSBJbnRyb2R1Y2luZyBHUFQtNiBTb2wgYW5kIEx1bmEgcGFnZS4","name":"Introducing GPT-6 Sol and Luna","notes":"Reported reasoning effort low. DeepSWE v1.1. Original long-horizon software-engineering tasks. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page."},{"benchmarkId":"deepswe-v1-1-1-1-percent","id":"deepswe-v1-1-1-1-percent:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtZWRpdW0uIERlZXBTV0UgdjEuMS4gT3JpZ2luYWwgbG9uZy1ob3Jpem9uIHNvZnR3YXJlLWVuZ2luZWVyaW5nIHRhc2tzLiBPcGVuQUktcHVibGlzaGVkIGNoYXJ0IG9uIHRoZSBJbnRyb2R1Y2luZyBHUFQtNiBTb2wgYW5kIEx1bmEgcGFnZS4","name":"Introducing GPT-6 Sol and Luna","notes":"Reported reasoning effort medium. DeepSWE v1.1. Original long-horizon software-engineering tasks. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page."},{"benchmarkId":"deepswe-v1-1-1-1-percent","id":"deepswe-v1-1-1-1-percent:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBoaWdoLiBEZWVwU1dFIHYxLjEuIE9yaWdpbmFsIGxvbmctaG9yaXpvbiBzb2Z0d2FyZS1lbmdpbmVlcmluZyB0YXNrcy4gT3BlbkFJLXB1Ymxpc2hlZCBjaGFydCBvbiB0aGUgSW50cm9kdWNpbmcgR1BULTYgU29sIGFuZCBMdW5hIHBhZ2Uu","name":"Introducing GPT-6 Sol and Luna","notes":"Reported reasoning effort high. DeepSWE v1.1. Original long-horizon software-engineering tasks. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page."},{"benchmarkId":"deepswe-v1-1-1-1-percent","id":"deepswe-v1-1-1-1-percent:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCB4aGlnaC4gRGVlcFNXRSB2MS4xLiBPcmlnaW5hbCBsb25nLWhvcml6b24gc29mdHdhcmUtZW5naW5lZXJpbmcgdGFza3MuIE9wZW5BSS1wdWJsaXNoZWQgY2hhcnQgb24gdGhlIEludHJvZHVjaW5nIEdQVC02IFNvbCBhbmQgTHVuYSBwYWdlLg","name":"Introducing GPT-6 Sol and Luna","notes":"Reported reasoning effort xhigh. DeepSWE v1.1. Original long-horizon software-engineering tasks. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page."},{"benchmarkId":"deepswe-v1-1-1-1-percent","id":"deepswe-v1-1-1-1-percent:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIERlZXBTV0UgdjEuMS4gT3JpZ2luYWwgbG9uZy1ob3Jpem9uIHNvZnR3YXJlLWVuZ2luZWVyaW5nIHRhc2tzLiBPcGVuQUktcHVibGlzaGVkIGNoYXJ0IG9uIHRoZSBJbnRyb2R1Y2luZyBHUFQtNiBTb2wgYW5kIEx1bmEgcGFnZS4","name":"Introducing GPT-6 Sol and Luna","notes":"Reported reasoning effort max. DeepSWE v1.1. Original long-horizon software-engineering tasks. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page."},{"benchmarkId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline","id":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBsb3cuIFBhcnRpYWwgcmV3YXJkIG9uIHRoZSBvZmZsaW5lIHNldCBmcm9tIHRoZSB2MjAyNi4wOC4wOCByZWxlYXNlLiBPcGVuQUktcHVibGlzaGVkIGNoYXJ0IG9uIHRoZSBJbnRyb2R1Y2luZyBHUFQtNiBTb2wgYW5kIEx1bmEgcGFnZS4","name":"Introducing GPT-6 Sol and Luna","notes":"Reported reasoning effort low. Partial reward on the offline set from the v2026.08.08 release. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page."},{"benchmarkId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline","id":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtZWRpdW0uIFBhcnRpYWwgcmV3YXJkIG9uIHRoZSBvZmZsaW5lIHNldCBmcm9tIHRoZSB2MjAyNi4wOC4wOCByZWxlYXNlLiBPcGVuQUktcHVibGlzaGVkIGNoYXJ0IG9uIHRoZSBJbnRyb2R1Y2luZyBHUFQtNiBTb2wgYW5kIEx1bmEgcGFnZS4","name":"Introducing GPT-6 Sol and Luna","notes":"Reported reasoning effort medium. Partial reward on the offline set from the v2026.08.08 release. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page."},{"benchmarkId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline","id":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBoaWdoLiBQYXJ0aWFsIHJld2FyZCBvbiB0aGUgb2ZmbGluZSBzZXQgZnJvbSB0aGUgdjIwMjYuMDguMDggcmVsZWFzZS4gT3BlbkFJLXB1Ymxpc2hlZCBjaGFydCBvbiB0aGUgSW50cm9kdWNpbmcgR1BULTYgU29sIGFuZCBMdW5hIHBhZ2Uu","name":"Introducing GPT-6 Sol and Luna","notes":"Reported reasoning effort high. Partial reward on the offline set from the v2026.08.08 release. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page."},{"benchmarkId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline","id":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCB4aGlnaC4gUGFydGlhbCByZXdhcmQgb24gdGhlIG9mZmxpbmUgc2V0IGZyb20gdGhlIHYyMDI2LjA4LjA4IHJlbGVhc2UuIE9wZW5BSS1wdWJsaXNoZWQgY2hhcnQgb24gdGhlIEludHJvZHVjaW5nIEdQVC02IFNvbCBhbmQgTHVuYSBwYWdlLg","name":"Introducing GPT-6 Sol and Luna","notes":"Reported reasoning effort xhigh. Partial reward on the offline set from the v2026.08.08 release. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page."},{"benchmarkId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline","id":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIFBhcnRpYWwgcmV3YXJkIG9uIHRoZSBvZmZsaW5lIHNldCBmcm9tIHRoZSB2MjAyNi4wOC4wOCByZWxlYXNlLiBPcGVuQUktcHVibGlzaGVkIGNoYXJ0IG9uIHRoZSBJbnRyb2R1Y2luZyBHUFQtNiBTb2wgYW5kIEx1bmEgcGFnZS4","name":"Introducing GPT-6 Sol and Luna","notes":"Reported reasoning effort max. Partial reward on the offline set from the v2026.08.08 release. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page."},{"benchmarkId":"deepswe-v1-1-1-1-percent","id":"deepswe-v1-1-1-1-percent:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBsb3cuIENvbXBsZXggc29mdHdhcmUtZW5naW5lZXJpbmcgdGFza3MgaW4gb3JpZ2luYWwgcmVhbCBjb2RlYmFzZXMuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","name":"Introducing GPT-6.1 Sol","notes":"Reported reasoning effort low. Complex software-engineering tasks in original real codebases. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration."},{"benchmarkId":"deepswe-v1-1-1-1-percent","id":"deepswe-v1-1-1-1-percent:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtZWRpdW0uIENvbXBsZXggc29mdHdhcmUtZW5naW5lZXJpbmcgdGFza3MgaW4gb3JpZ2luYWwgcmVhbCBjb2RlYmFzZXMuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","name":"Introducing GPT-6.1 Sol","notes":"Reported reasoning effort medium. Complex software-engineering tasks in original real codebases. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration."},{"benchmarkId":"deepswe-v1-1-1-1-percent","id":"deepswe-v1-1-1-1-percent:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBoaWdoLiBDb21wbGV4IHNvZnR3YXJlLWVuZ2luZWVyaW5nIHRhc2tzIGluIG9yaWdpbmFsIHJlYWwgY29kZWJhc2VzLiBPcGVuQUkgcmVzZWFyY2ggZW52aXJvbm1lbnQgb3IgQVBJOyBzYW1lIG5hbWVkIG1ldHJpYyBhbmQgZXZhbHVhdGlvbiBjaGFydC4gTm8gZmFsbGJhY2sgaXMgcmVwb3J0ZWQgZm9yIHRoaXMgT3BlbkFJIGNvbmZpZ3VyYXRpb24u","name":"Introducing GPT-6.1 Sol","notes":"Reported reasoning effort high. Complex software-engineering tasks in original real codebases. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration."},{"benchmarkId":"deepswe-v1-1-1-1-percent","id":"deepswe-v1-1-1-1-percent:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCB4aGlnaC4gQ29tcGxleCBzb2Z0d2FyZS1lbmdpbmVlcmluZyB0YXNrcyBpbiBvcmlnaW5hbCByZWFsIGNvZGViYXNlcy4gT3BlbkFJIHJlc2VhcmNoIGVudmlyb25tZW50IG9yIEFQSTsgc2FtZSBuYW1lZCBtZXRyaWMgYW5kIGV2YWx1YXRpb24gY2hhcnQuIE5vIGZhbGxiYWNrIGlzIHJlcG9ydGVkIGZvciB0aGlzIE9wZW5BSSBjb25maWd1cmF0aW9uLg","name":"Introducing GPT-6.1 Sol","notes":"Reported reasoning effort xhigh. Complex software-engineering tasks in original real codebases. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration."},{"benchmarkId":"deepswe-v1-1-1-1-percent","id":"deepswe-v1-1-1-1-percent:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIENvbXBsZXggc29mdHdhcmUtZW5naW5lZXJpbmcgdGFza3MgaW4gb3JpZ2luYWwgcmVhbCBjb2RlYmFzZXMuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","name":"Introducing GPT-6.1 Sol","notes":"Reported reasoning effort max. Complex software-engineering tasks in original real codebases. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration."},{"benchmarkId":"gdp-pdf-not-specified","id":"gdp-pdf-not-specified:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBsb3cuIFByb2Zlc3Npb25hbCBxdWVzdGlvbnMgYWJvdXQgY29tcGxleCBQREZzIGFjcm9zcyB0ZW4gcHJvZmVzc2lvbmFsIGRvbWFpbnMuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","name":"Introducing GPT-6.1 Sol","notes":"Reported reasoning effort low. Professional questions about complex PDFs across ten professional domains. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration."},{"benchmarkId":"gdp-pdf-not-specified","id":"gdp-pdf-not-specified:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtZWRpdW0uIFByb2Zlc3Npb25hbCBxdWVzdGlvbnMgYWJvdXQgY29tcGxleCBQREZzIGFjcm9zcyB0ZW4gcHJvZmVzc2lvbmFsIGRvbWFpbnMuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","name":"Introducing GPT-6.1 Sol","notes":"Reported reasoning effort medium. Professional questions about complex PDFs across ten professional domains. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration."},{"benchmarkId":"gdp-pdf-not-specified","id":"gdp-pdf-not-specified:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBoaWdoLiBQcm9mZXNzaW9uYWwgcXVlc3Rpb25zIGFib3V0IGNvbXBsZXggUERGcyBhY3Jvc3MgdGVuIHByb2Zlc3Npb25hbCBkb21haW5zLiBPcGVuQUkgcmVzZWFyY2ggZW52aXJvbm1lbnQgb3IgQVBJOyBzYW1lIG5hbWVkIG1ldHJpYyBhbmQgZXZhbHVhdGlvbiBjaGFydC4gTm8gZmFsbGJhY2sgaXMgcmVwb3J0ZWQgZm9yIHRoaXMgT3BlbkFJIGNvbmZpZ3VyYXRpb24u","name":"Introducing GPT-6.1 Sol","notes":"Reported reasoning effort high. Professional questions about complex PDFs across ten professional domains. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration."},{"benchmarkId":"gdp-pdf-not-specified","id":"gdp-pdf-not-specified:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCB4aGlnaC4gUHJvZmVzc2lvbmFsIHF1ZXN0aW9ucyBhYm91dCBjb21wbGV4IFBERnMgYWNyb3NzIHRlbiBwcm9mZXNzaW9uYWwgZG9tYWlucy4gT3BlbkFJIHJlc2VhcmNoIGVudmlyb25tZW50IG9yIEFQSTsgc2FtZSBuYW1lZCBtZXRyaWMgYW5kIGV2YWx1YXRpb24gY2hhcnQuIE5vIGZhbGxiYWNrIGlzIHJlcG9ydGVkIGZvciB0aGlzIE9wZW5BSSBjb25maWd1cmF0aW9uLg","name":"Introducing GPT-6.1 Sol","notes":"Reported reasoning effort xhigh. Professional questions about complex PDFs across ten professional domains. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration."},{"benchmarkId":"gdp-pdf-not-specified","id":"gdp-pdf-not-specified:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIFByb2Zlc3Npb25hbCBxdWVzdGlvbnMgYWJvdXQgY29tcGxleCBQREZzIGFjcm9zcyB0ZW4gcHJvZmVzc2lvbmFsIGRvbWFpbnMuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","name":"Introducing GPT-6.1 Sol","notes":"Reported reasoning effort max. Professional questions about complex PDFs across ten professional domains. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration."},{"benchmarkId":"gdp-pdf-not-specified","id":"gdp-pdf-not-specified:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBsb3cuIFByb2Zlc3Npb25hbCBxdWVzdGlvbnMgYWJvdXQgY29tcGxleCBQREZzIGFjcm9zcyB0ZW4gcHJvZmVzc2lvbmFsIGRvbWFpbnMuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBSZXBvcnRlZCBjb21iaW5lZCBzZXR1cDogT3B1cyA1LjUgdy8gZmFsbGJhY2tzLiBGYWxsYmFjayBiZWhhdmlvciBtdXN0IG5vdCBiZSBhdHRyaWJ1dGVkIHRvIHRoZSBiYXNlIG1vZGVsIGFsb25lLg","name":"Introducing GPT-6.1 Sol","notes":"Reported reasoning effort low. Professional questions about complex PDFs across ten professional domains. OpenAI research environment or API; same named metric and evaluation chart. Reported combined setup: Opus 5.5 w/ fallbacks. Fallback behavior must not be attributed to the base model alone."},{"benchmarkId":"gdp-pdf-not-specified","id":"gdp-pdf-not-specified:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtZWRpdW0uIFByb2Zlc3Npb25hbCBxdWVzdGlvbnMgYWJvdXQgY29tcGxleCBQREZzIGFjcm9zcyB0ZW4gcHJvZmVzc2lvbmFsIGRvbWFpbnMuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBSZXBvcnRlZCBjb21iaW5lZCBzZXR1cDogT3B1cyA1LjUgdy8gZmFsbGJhY2tzLiBGYWxsYmFjayBiZWhhdmlvciBtdXN0IG5vdCBiZSBhdHRyaWJ1dGVkIHRvIHRoZSBiYXNlIG1vZGVsIGFsb25lLg","name":"Introducing GPT-6.1 Sol","notes":"Reported reasoning effort medium. Professional questions about complex PDFs across ten professional domains. OpenAI research environment or API; same named metric and evaluation chart. Reported combined setup: Opus 5.5 w/ fallbacks. Fallback behavior must not be attributed to the base model alone."},{"benchmarkId":"gdp-pdf-not-specified","id":"gdp-pdf-not-specified:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBoaWdoLiBQcm9mZXNzaW9uYWwgcXVlc3Rpb25zIGFib3V0IGNvbXBsZXggUERGcyBhY3Jvc3MgdGVuIHByb2Zlc3Npb25hbCBkb21haW5zLiBPcGVuQUkgcmVzZWFyY2ggZW52aXJvbm1lbnQgb3IgQVBJOyBzYW1lIG5hbWVkIG1ldHJpYyBhbmQgZXZhbHVhdGlvbiBjaGFydC4gUmVwb3J0ZWQgY29tYmluZWQgc2V0dXA6IE9wdXMgNS41IHcvIGZhbGxiYWNrcy4gRmFsbGJhY2sgYmVoYXZpb3IgbXVzdCBub3QgYmUgYXR0cmlidXRlZCB0byB0aGUgYmFzZSBtb2RlbCBhbG9uZS4","name":"Introducing GPT-6.1 Sol","notes":"Reported reasoning effort high. Professional questions about complex PDFs across ten professional domains. OpenAI research environment or API; same named metric and evaluation chart. Reported combined setup: Opus 5.5 w/ fallbacks. Fallback behavior must not be attributed to the base model alone."},{"benchmarkId":"gdp-pdf-not-specified","id":"gdp-pdf-not-specified:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCB4aGlnaC4gUHJvZmVzc2lvbmFsIHF1ZXN0aW9ucyBhYm91dCBjb21wbGV4IFBERnMgYWNyb3NzIHRlbiBwcm9mZXNzaW9uYWwgZG9tYWlucy4gT3BlbkFJIHJlc2VhcmNoIGVudmlyb25tZW50IG9yIEFQSTsgc2FtZSBuYW1lZCBtZXRyaWMgYW5kIGV2YWx1YXRpb24gY2hhcnQuIFJlcG9ydGVkIGNvbWJpbmVkIHNldHVwOiBPcHVzIDUuNSB3LyBmYWxsYmFja3MuIEZhbGxiYWNrIGJlaGF2aW9yIG11c3Qgbm90IGJlIGF0dHJpYnV0ZWQgdG8gdGhlIGJhc2UgbW9kZWwgYWxvbmUu","name":"Introducing GPT-6.1 Sol","notes":"Reported reasoning effort xhigh. Professional questions about complex PDFs across ten professional domains. OpenAI research environment or API; same named metric and evaluation chart. Reported combined setup: Opus 5.5 w/ fallbacks. Fallback behavior must not be attributed to the base model alone."},{"benchmarkId":"gdp-pdf-not-specified","id":"gdp-pdf-not-specified:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIFByb2Zlc3Npb25hbCBxdWVzdGlvbnMgYWJvdXQgY29tcGxleCBQREZzIGFjcm9zcyB0ZW4gcHJvZmVzc2lvbmFsIGRvbWFpbnMuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBSZXBvcnRlZCBjb21iaW5lZCBzZXR1cDogT3B1cyA1LjUgdy8gZmFsbGJhY2tzLiBGYWxsYmFjayBiZWhhdmlvciBtdXN0IG5vdCBiZSBhdHRyaWJ1dGVkIHRvIHRoZSBiYXNlIG1vZGVsIGFsb25lLg","name":"Introducing GPT-6.1 Sol","notes":"Reported reasoning effort max. Professional questions about complex PDFs across ten professional domains. OpenAI research environment or API; same named metric and evaluation chart. Reported combined setup: Opus 5.5 w/ fallbacks. Fallback behavior must not be attributed to the base model alone."},{"benchmarkId":"automationbench-1-0-6-percent","id":"automationbench-1-0-6-percent:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIEVuZC10by1lbmQgYnVzaW5lc3Mgd29ya2Zsb3dzIHVzaW5nIDQ3IHRvb2xzIGFjcm9zcyBzaXggYnVzaW5lc3MgZnVuY3Rpb25zLiBPcGVuQUkgcmVzZWFyY2ggZW52aXJvbm1lbnQgb3IgQVBJOyBzYW1lIG5hbWVkIG1ldHJpYyBhbmQgZXZhbHVhdGlvbiBjaGFydC4gUmVwb3J0ZWQgY29tYmluZWQgc2V0dXA6IEZhYmxlIDUuMSB3LyBPcHVzIDUgZmFsbGJhY2suIEZhbGxiYWNrIGJlaGF2aW9yIG11c3Qgbm90IGJlIGF0dHJpYnV0ZWQgdG8gdGhlIGJhc2UgbW9kZWwgYWxvbmUu","name":"Introducing GPT-6.1 Sol","notes":"Reported reasoning effort max. End-to-end business workflows using 47 tools across six business functions. OpenAI research environment or API; same named metric and evaluation chart. Reported combined setup: Fable 5.1 w/ Opus 5 fallback. Fallback behavior must not be attributed to the base model alone."},{"benchmarkId":"automationbench-1-0-6-percent","id":"automationbench-1-0-6-percent:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBsb3cuIEVuZC10by1lbmQgYnVzaW5lc3Mgd29ya2Zsb3dzIHVzaW5nIDQ3IHRvb2xzIGFjcm9zcyBzaXggYnVzaW5lc3MgZnVuY3Rpb25zLiBPcGVuQUkgcmVzZWFyY2ggZW52aXJvbm1lbnQgb3IgQVBJOyBzYW1lIG5hbWVkIG1ldHJpYyBhbmQgZXZhbHVhdGlvbiBjaGFydC4gTm8gZmFsbGJhY2sgaXMgcmVwb3J0ZWQgZm9yIHRoaXMgT3BlbkFJIGNvbmZpZ3VyYXRpb24u","name":"Introducing GPT-6.1 Sol","notes":"Reported reasoning effort low. End-to-end business workflows using 47 tools across six business functions. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration."},{"benchmarkId":"automationbench-1-0-6-percent","id":"automationbench-1-0-6-percent:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtZWRpdW0uIEVuZC10by1lbmQgYnVzaW5lc3Mgd29ya2Zsb3dzIHVzaW5nIDQ3IHRvb2xzIGFjcm9zcyBzaXggYnVzaW5lc3MgZnVuY3Rpb25zLiBPcGVuQUkgcmVzZWFyY2ggZW52aXJvbm1lbnQgb3IgQVBJOyBzYW1lIG5hbWVkIG1ldHJpYyBhbmQgZXZhbHVhdGlvbiBjaGFydC4gTm8gZmFsbGJhY2sgaXMgcmVwb3J0ZWQgZm9yIHRoaXMgT3BlbkFJIGNvbmZpZ3VyYXRpb24u","name":"Introducing GPT-6.1 Sol","notes":"Reported reasoning effort medium. End-to-end business workflows using 47 tools across six business functions. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration."},{"benchmarkId":"automationbench-1-0-6-percent","id":"automationbench-1-0-6-percent:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBoaWdoLiBFbmQtdG8tZW5kIGJ1c2luZXNzIHdvcmtmbG93cyB1c2luZyA0NyB0b29scyBhY3Jvc3Mgc2l4IGJ1c2luZXNzIGZ1bmN0aW9ucy4gT3BlbkFJIHJlc2VhcmNoIGVudmlyb25tZW50IG9yIEFQSTsgc2FtZSBuYW1lZCBtZXRyaWMgYW5kIGV2YWx1YXRpb24gY2hhcnQuIE5vIGZhbGxiYWNrIGlzIHJlcG9ydGVkIGZvciB0aGlzIE9wZW5BSSBjb25maWd1cmF0aW9uLg","name":"Introducing GPT-6.1 Sol","notes":"Reported reasoning effort high. End-to-end business workflows using 47 tools across six business functions. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration."},{"benchmarkId":"automationbench-1-0-6-percent","id":"automationbench-1-0-6-percent:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCB4aGlnaC4gRW5kLXRvLWVuZCBidXNpbmVzcyB3b3JrZmxvd3MgdXNpbmcgNDcgdG9vbHMgYWNyb3NzIHNpeCBidXNpbmVzcyBmdW5jdGlvbnMuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","name":"Introducing GPT-6.1 Sol","notes":"Reported reasoning effort xhigh. End-to-end business workflows using 47 tools across six business functions. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration."},{"benchmarkId":"automationbench-1-0-6-percent","id":"automationbench-1-0-6-percent:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIEVuZC10by1lbmQgYnVzaW5lc3Mgd29ya2Zsb3dzIHVzaW5nIDQ3IHRvb2xzIGFjcm9zcyBzaXggYnVzaW5lc3MgZnVuY3Rpb25zLiBPcGVuQUkgcmVzZWFyY2ggZW52aXJvbm1lbnQgb3IgQVBJOyBzYW1lIG5hbWVkIG1ldHJpYyBhbmQgZXZhbHVhdGlvbiBjaGFydC4gTm8gZmFsbGJhY2sgaXMgcmVwb3J0ZWQgZm9yIHRoaXMgT3BlbkFJIGNvbmZpZ3VyYXRpb24u","name":"Introducing GPT-6.1 Sol","notes":"Reported reasoning effort max. End-to-end business workflows using 47 tools across six business functions. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration."},{"benchmarkId":"automationbench-1-0-6-percent","id":"automationbench-1-0-6-percent:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBsb3cuIEVuZC10by1lbmQgYnVzaW5lc3Mgd29ya2Zsb3dzIHVzaW5nIDQ3IHRvb2xzIGFjcm9zcyBzaXggYnVzaW5lc3MgZnVuY3Rpb25zLiBPcGVuQUkgcmVzZWFyY2ggZW52aXJvbm1lbnQgb3IgQVBJOyBzYW1lIG5hbWVkIG1ldHJpYyBhbmQgZXZhbHVhdGlvbiBjaGFydC4gUmVwb3J0ZWQgY29tYmluZWQgc2V0dXA6IE9wdXMgNS41IHcvIGZhbGxiYWNrcy4gRmFsbGJhY2sgYmVoYXZpb3IgbXVzdCBub3QgYmUgYXR0cmlidXRlZCB0byB0aGUgYmFzZSBtb2RlbCBhbG9uZS4","name":"Introducing GPT-6.1 Sol","notes":"Reported reasoning effort low. End-to-end business workflows using 47 tools across six business functions. OpenAI research environment or API; same named metric and evaluation chart. Reported combined setup: Opus 5.5 w/ fallbacks. Fallback behavior must not be attributed to the base model alone."},{"benchmarkId":"automationbench-1-0-6-percent","id":"automationbench-1-0-6-percent:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtZWRpdW0uIEVuZC10by1lbmQgYnVzaW5lc3Mgd29ya2Zsb3dzIHVzaW5nIDQ3IHRvb2xzIGFjcm9zcyBzaXggYnVzaW5lc3MgZnVuY3Rpb25zLiBPcGVuQUkgcmVzZWFyY2ggZW52aXJvbm1lbnQgb3IgQVBJOyBzYW1lIG5hbWVkIG1ldHJpYyBhbmQgZXZhbHVhdGlvbiBjaGFydC4gUmVwb3J0ZWQgY29tYmluZWQgc2V0dXA6IE9wdXMgNS41IHcvIGZhbGxiYWNrcy4gRmFsbGJhY2sgYmVoYXZpb3IgbXVzdCBub3QgYmUgYXR0cmlidXRlZCB0byB0aGUgYmFzZSBtb2RlbCBhbG9uZS4","name":"Introducing GPT-6.1 Sol","notes":"Reported reasoning effort medium. End-to-end business workflows using 47 tools across six business functions. OpenAI research environment or API; same named metric and evaluation chart. Reported combined setup: Opus 5.5 w/ fallbacks. Fallback behavior must not be attributed to the base model alone."},{"benchmarkId":"automationbench-1-0-6-percent","id":"automationbench-1-0-6-percent:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBoaWdoLiBFbmQtdG8tZW5kIGJ1c2luZXNzIHdvcmtmbG93cyB1c2luZyA0NyB0b29scyBhY3Jvc3Mgc2l4IGJ1c2luZXNzIGZ1bmN0aW9ucy4gT3BlbkFJIHJlc2VhcmNoIGVudmlyb25tZW50IG9yIEFQSTsgc2FtZSBuYW1lZCBtZXRyaWMgYW5kIGV2YWx1YXRpb24gY2hhcnQuIFJlcG9ydGVkIGNvbWJpbmVkIHNldHVwOiBPcHVzIDUuNSB3LyBmYWxsYmFja3MuIEZhbGxiYWNrIGJlaGF2aW9yIG11c3Qgbm90IGJlIGF0dHJpYnV0ZWQgdG8gdGhlIGJhc2UgbW9kZWwgYWxvbmUu","name":"Introducing GPT-6.1 Sol","notes":"Reported reasoning effort high. End-to-end business workflows using 47 tools across six business functions. OpenAI research environment or API; same named metric and evaluation chart. Reported combined setup: Opus 5.5 w/ fallbacks. Fallback behavior must not be attributed to the base model alone."},{"benchmarkId":"automationbench-1-0-6-percent","id":"automationbench-1-0-6-percent:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCB4aGlnaC4gRW5kLXRvLWVuZCBidXNpbmVzcyB3b3JrZmxvd3MgdXNpbmcgNDcgdG9vbHMgYWNyb3NzIHNpeCBidXNpbmVzcyBmdW5jdGlvbnMuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBSZXBvcnRlZCBjb21iaW5lZCBzZXR1cDogT3B1cyA1LjUgdy8gZmFsbGJhY2tzLiBGYWxsYmFjayBiZWhhdmlvciBtdXN0IG5vdCBiZSBhdHRyaWJ1dGVkIHRvIHRoZSBiYXNlIG1vZGVsIGFsb25lLg","name":"Introducing GPT-6.1 Sol","notes":"Reported reasoning effort xhigh. End-to-end business workflows using 47 tools across six business functions. OpenAI research environment or API; same named metric and evaluation chart. Reported combined setup: Opus 5.5 w/ fallbacks. Fallback behavior must not be attributed to the base model alone."},{"benchmarkId":"automationbench-1-0-6-percent","id":"automationbench-1-0-6-percent:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIEVuZC10by1lbmQgYnVzaW5lc3Mgd29ya2Zsb3dzIHVzaW5nIDQ3IHRvb2xzIGFjcm9zcyBzaXggYnVzaW5lc3MgZnVuY3Rpb25zLiBPcGVuQUkgcmVzZWFyY2ggZW52aXJvbm1lbnQgb3IgQVBJOyBzYW1lIG5hbWVkIG1ldHJpYyBhbmQgZXZhbHVhdGlvbiBjaGFydC4gUmVwb3J0ZWQgY29tYmluZWQgc2V0dXA6IE9wdXMgNS41IHcvIGZhbGxiYWNrcy4gRmFsbGJhY2sgYmVoYXZpb3IgbXVzdCBub3QgYmUgYXR0cmlidXRlZCB0byB0aGUgYmFzZSBtb2RlbCBhbG9uZS4","name":"Introducing GPT-6.1 Sol","notes":"Reported reasoning effort max. End-to-end business workflows using 47 tools across six business functions. OpenAI research environment or API; same named metric and evaluation chart. Reported combined setup: Opus 5.5 w/ fallbacks. Fallback behavior must not be attributed to the base model alone."},{"benchmarkId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline","id":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBsb3cuIE9mZmxpbmUgc2V0IGZyb20gdjIwMjYuMDguMDg7IHBhcnRpYWwgcmV3YXJkLCBub3QgYmluYXJ5IGZ1bGwtdGFzayBzdWNjZXNzIG9yIE9TV29ybGQgVmVyaWZpZWQuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","name":"Introducing GPT-6.1 Sol","notes":"Reported reasoning effort low. Offline set from v2026.08.08; partial reward, not binary full-task success or OSWorld Verified. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration."},{"benchmarkId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline","id":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtZWRpdW0uIE9mZmxpbmUgc2V0IGZyb20gdjIwMjYuMDguMDg7IHBhcnRpYWwgcmV3YXJkLCBub3QgYmluYXJ5IGZ1bGwtdGFzayBzdWNjZXNzIG9yIE9TV29ybGQgVmVyaWZpZWQuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","name":"Introducing GPT-6.1 Sol","notes":"Reported reasoning effort medium. Offline set from v2026.08.08; partial reward, not binary full-task success or OSWorld Verified. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration."},{"benchmarkId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline","id":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBoaWdoLiBPZmZsaW5lIHNldCBmcm9tIHYyMDI2LjA4LjA4OyBwYXJ0aWFsIHJld2FyZCwgbm90IGJpbmFyeSBmdWxsLXRhc2sgc3VjY2VzcyBvciBPU1dvcmxkIFZlcmlmaWVkLiBPcGVuQUkgcmVzZWFyY2ggZW52aXJvbm1lbnQgb3IgQVBJOyBzYW1lIG5hbWVkIG1ldHJpYyBhbmQgZXZhbHVhdGlvbiBjaGFydC4gTm8gZmFsbGJhY2sgaXMgcmVwb3J0ZWQgZm9yIHRoaXMgT3BlbkFJIGNvbmZpZ3VyYXRpb24u","name":"Introducing GPT-6.1 Sol","notes":"Reported reasoning effort high. Offline set from v2026.08.08; partial reward, not binary full-task success or OSWorld Verified. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration."},{"benchmarkId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline","id":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCB4aGlnaC4gT2ZmbGluZSBzZXQgZnJvbSB2MjAyNi4wOC4wODsgcGFydGlhbCByZXdhcmQsIG5vdCBiaW5hcnkgZnVsbC10YXNrIHN1Y2Nlc3Mgb3IgT1NXb3JsZCBWZXJpZmllZC4gT3BlbkFJIHJlc2VhcmNoIGVudmlyb25tZW50IG9yIEFQSTsgc2FtZSBuYW1lZCBtZXRyaWMgYW5kIGV2YWx1YXRpb24gY2hhcnQuIE5vIGZhbGxiYWNrIGlzIHJlcG9ydGVkIGZvciB0aGlzIE9wZW5BSSBjb25maWd1cmF0aW9uLg","name":"Introducing GPT-6.1 Sol","notes":"Reported reasoning effort xhigh. Offline set from v2026.08.08; partial reward, not binary full-task success or OSWorld Verified. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration."},{"benchmarkId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline","id":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIE9mZmxpbmUgc2V0IGZyb20gdjIwMjYuMDguMDg7IHBhcnRpYWwgcmV3YXJkLCBub3QgYmluYXJ5IGZ1bGwtdGFzayBzdWNjZXNzIG9yIE9TV29ybGQgVmVyaWZpZWQuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","name":"Introducing GPT-6.1 Sol","notes":"Reported reasoning effort max. Offline set from v2026.08.08; partial reward, not binary full-task success or OSWorld Verified. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration."},{"benchmarkId":"terminal-bench-science-0-1","id":"terminal-bench-science-0-1:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIFNjaWVudGlmaWMgcmVzZWFyY2ggd29ya2Zsb3dzIHVzaW5nIGNvZGUgYW5kIHRlcm1pbmFsIHRvb2xzLCBpbmNsdWRpbmcgYW5hbHlzaXMsIHNpbXVsYXRpb24gYW5kIG1vZGVsIGZpdHRpbmcuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBSZXBvcnRlZCBjb21iaW5lZCBzZXR1cDogT3B1cyA1LjUgdy8gZmFsbGJhY2tzLiBGYWxsYmFjayBiZWhhdmlvciBtdXN0IG5vdCBiZSBhdHRyaWJ1dGVkIHRvIHRoZSBiYXNlIG1vZGVsIGFsb25lLg","name":"Introducing GPT-6.1 Sol","notes":"Reported reasoning effort max. Scientific research workflows using code and terminal tools, including analysis, simulation and model fitting. OpenAI research environment or API; same named metric and evaluation chart. Reported combined setup: Opus 5.5 w/ fallbacks. Fallback behavior must not be attributed to the base model alone."},{"benchmarkId":"terminal-bench-science-0-1","id":"terminal-bench-science-0-1:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBsb3cuIFNjaWVudGlmaWMgcmVzZWFyY2ggd29ya2Zsb3dzIHVzaW5nIGNvZGUgYW5kIHRlcm1pbmFsIHRvb2xzLCBpbmNsdWRpbmcgYW5hbHlzaXMsIHNpbXVsYXRpb24gYW5kIG1vZGVsIGZpdHRpbmcuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","name":"Introducing GPT-6.1 Sol","notes":"Reported reasoning effort low. Scientific research workflows using code and terminal tools, including analysis, simulation and model fitting. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration."},{"benchmarkId":"terminal-bench-science-0-1","id":"terminal-bench-science-0-1:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtZWRpdW0uIFNjaWVudGlmaWMgcmVzZWFyY2ggd29ya2Zsb3dzIHVzaW5nIGNvZGUgYW5kIHRlcm1pbmFsIHRvb2xzLCBpbmNsdWRpbmcgYW5hbHlzaXMsIHNpbXVsYXRpb24gYW5kIG1vZGVsIGZpdHRpbmcuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","name":"Introducing GPT-6.1 Sol","notes":"Reported reasoning effort medium. Scientific research workflows using code and terminal tools, including analysis, simulation and model fitting. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration."},{"benchmarkId":"terminal-bench-science-0-1","id":"terminal-bench-science-0-1:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBoaWdoLiBTY2llbnRpZmljIHJlc2VhcmNoIHdvcmtmbG93cyB1c2luZyBjb2RlIGFuZCB0ZXJtaW5hbCB0b29scywgaW5jbHVkaW5nIGFuYWx5c2lzLCBzaW11bGF0aW9uIGFuZCBtb2RlbCBmaXR0aW5nLiBPcGVuQUkgcmVzZWFyY2ggZW52aXJvbm1lbnQgb3IgQVBJOyBzYW1lIG5hbWVkIG1ldHJpYyBhbmQgZXZhbHVhdGlvbiBjaGFydC4gTm8gZmFsbGJhY2sgaXMgcmVwb3J0ZWQgZm9yIHRoaXMgT3BlbkFJIGNvbmZpZ3VyYXRpb24u","name":"Introducing GPT-6.1 Sol","notes":"Reported reasoning effort high. Scientific research workflows using code and terminal tools, including analysis, simulation and model fitting. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration."},{"benchmarkId":"terminal-bench-science-0-1","id":"terminal-bench-science-0-1:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCB4aGlnaC4gU2NpZW50aWZpYyByZXNlYXJjaCB3b3JrZmxvd3MgdXNpbmcgY29kZSBhbmQgdGVybWluYWwgdG9vbHMsIGluY2x1ZGluZyBhbmFseXNpcywgc2ltdWxhdGlvbiBhbmQgbW9kZWwgZml0dGluZy4gT3BlbkFJIHJlc2VhcmNoIGVudmlyb25tZW50IG9yIEFQSTsgc2FtZSBuYW1lZCBtZXRyaWMgYW5kIGV2YWx1YXRpb24gY2hhcnQuIE5vIGZhbGxiYWNrIGlzIHJlcG9ydGVkIGZvciB0aGlzIE9wZW5BSSBjb25maWd1cmF0aW9uLg","name":"Introducing GPT-6.1 Sol","notes":"Reported reasoning effort xhigh. Scientific research workflows using code and terminal tools, including analysis, simulation and model fitting. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration."},{"benchmarkId":"terminal-bench-science-0-1","id":"terminal-bench-science-0-1:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIFNjaWVudGlmaWMgcmVzZWFyY2ggd29ya2Zsb3dzIHVzaW5nIGNvZGUgYW5kIHRlcm1pbmFsIHRvb2xzLCBpbmNsdWRpbmcgYW5hbHlzaXMsIHNpbXVsYXRpb24gYW5kIG1vZGVsIGZpdHRpbmcuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","name":"Introducing GPT-6.1 Sol","notes":"Reported reasoning effort max. Scientific research workflows using code and terminal tools, including analysis, simulation and model fitting. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration."},{"benchmarkId":"factual-error-rate-on-difficult-prompts-sol-6-1-launch-user-flagged-conversations","id":"factual-error-rate-on-difficult-prompts-sol-6-1-launch-user-flagged-conversations:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBsb3cuIFNoYXJlIG9mIGFuc3dlcnMgd2l0aCBhdCBsZWFzdCBvbmUgZmFjdHVhbCBlcnJvciBvbiBkZS1pZGVudGlmaWVkIGNvbnZlcnNhdGlvbnMgd2hlcmUgdXNlcnMgZmxhZ2dlZCBhbiBlYXJsaWVyIG1vZGVsIGVycm9yOyBkZWxpYmVyYXRlbHkgZGlmZmljdWx0LCBub3QgcmVwcmVzZW50YXRpdmUgb2YgdHlwaWNhbCB1c2FnZS4gT3BlbkFJIHJlc2VhcmNoIGVudmlyb25tZW50IG9yIEFQSTsgc2FtZSBuYW1lZCBtZXRyaWMgYW5kIGV2YWx1YXRpb24gY2hhcnQuIE5vIGZhbGxiYWNrIGlzIHJlcG9ydGVkIGZvciB0aGlzIE9wZW5BSSBjb25maWd1cmF0aW9uLg","name":"Introducing GPT-6.1 Sol","notes":"Reported reasoning effort low. Share of answers with at least one factual error on de-identified conversations where users flagged an earlier model error; deliberately difficult, not representative of typical usage. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration."},{"benchmarkId":"factual-error-rate-on-difficult-prompts-sol-6-1-launch-user-flagged-conversations","id":"factual-error-rate-on-difficult-prompts-sol-6-1-launch-user-flagged-conversations:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtZWRpdW0uIFNoYXJlIG9mIGFuc3dlcnMgd2l0aCBhdCBsZWFzdCBvbmUgZmFjdHVhbCBlcnJvciBvbiBkZS1pZGVudGlmaWVkIGNvbnZlcnNhdGlvbnMgd2hlcmUgdXNlcnMgZmxhZ2dlZCBhbiBlYXJsaWVyIG1vZGVsIGVycm9yOyBkZWxpYmVyYXRlbHkgZGlmZmljdWx0LCBub3QgcmVwcmVzZW50YXRpdmUgb2YgdHlwaWNhbCB1c2FnZS4gT3BlbkFJIHJlc2VhcmNoIGVudmlyb25tZW50IG9yIEFQSTsgc2FtZSBuYW1lZCBtZXRyaWMgYW5kIGV2YWx1YXRpb24gY2hhcnQuIE5vIGZhbGxiYWNrIGlzIHJlcG9ydGVkIGZvciB0aGlzIE9wZW5BSSBjb25maWd1cmF0aW9uLg","name":"Introducing GPT-6.1 Sol","notes":"Reported reasoning effort medium. Share of answers with at least one factual error on de-identified conversations where users flagged an earlier model error; deliberately difficult, not representative of typical usage. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration."},{"benchmarkId":"factual-error-rate-on-difficult-prompts-sol-6-1-launch-user-flagged-conversations","id":"factual-error-rate-on-difficult-prompts-sol-6-1-launch-user-flagged-conversations:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBoaWdoLiBTaGFyZSBvZiBhbnN3ZXJzIHdpdGggYXQgbGVhc3Qgb25lIGZhY3R1YWwgZXJyb3Igb24gZGUtaWRlbnRpZmllZCBjb252ZXJzYXRpb25zIHdoZXJlIHVzZXJzIGZsYWdnZWQgYW4gZWFybGllciBtb2RlbCBlcnJvcjsgZGVsaWJlcmF0ZWx5IGRpZmZpY3VsdCwgbm90IHJlcHJlc2VudGF0aXZlIG9mIHR5cGljYWwgdXNhZ2UuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","name":"Introducing GPT-6.1 Sol","notes":"Reported reasoning effort high. Share of answers with at least one factual error on de-identified conversations where users flagged an earlier model error; deliberately difficult, not representative of typical usage. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration."},{"benchmarkId":"factual-error-rate-on-difficult-prompts-sol-6-1-launch-user-flagged-conversations","id":"factual-error-rate-on-difficult-prompts-sol-6-1-launch-user-flagged-conversations:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCB4aGlnaC4gU2hhcmUgb2YgYW5zd2VycyB3aXRoIGF0IGxlYXN0IG9uZSBmYWN0dWFsIGVycm9yIG9uIGRlLWlkZW50aWZpZWQgY29udmVyc2F0aW9ucyB3aGVyZSB1c2VycyBmbGFnZ2VkIGFuIGVhcmxpZXIgbW9kZWwgZXJyb3I7IGRlbGliZXJhdGVseSBkaWZmaWN1bHQsIG5vdCByZXByZXNlbnRhdGl2ZSBvZiB0eXBpY2FsIHVzYWdlLiBPcGVuQUkgcmVzZWFyY2ggZW52aXJvbm1lbnQgb3IgQVBJOyBzYW1lIG5hbWVkIG1ldHJpYyBhbmQgZXZhbHVhdGlvbiBjaGFydC4gTm8gZmFsbGJhY2sgaXMgcmVwb3J0ZWQgZm9yIHRoaXMgT3BlbkFJIGNvbmZpZ3VyYXRpb24u","name":"Introducing GPT-6.1 Sol","notes":"Reported reasoning effort xhigh. Share of answers with at least one factual error on de-identified conversations where users flagged an earlier model error; deliberately difficult, not representative of typical usage. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration."},{"benchmarkId":"factual-error-rate-on-difficult-prompts-sol-6-1-launch-user-flagged-conversations","id":"factual-error-rate-on-difficult-prompts-sol-6-1-launch-user-flagged-conversations:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIFNoYXJlIG9mIGFuc3dlcnMgd2l0aCBhdCBsZWFzdCBvbmUgZmFjdHVhbCBlcnJvciBvbiBkZS1pZGVudGlmaWVkIGNvbnZlcnNhdGlvbnMgd2hlcmUgdXNlcnMgZmxhZ2dlZCBhbiBlYXJsaWVyIG1vZGVsIGVycm9yOyBkZWxpYmVyYXRlbHkgZGlmZmljdWx0LCBub3QgcmVwcmVzZW50YXRpdmUgb2YgdHlwaWNhbCB1c2FnZS4gT3BlbkFJIHJlc2VhcmNoIGVudmlyb25tZW50IG9yIEFQSTsgc2FtZSBuYW1lZCBtZXRyaWMgYW5kIGV2YWx1YXRpb24gY2hhcnQuIE5vIGZhbGxiYWNrIGlzIHJlcG9ydGVkIGZvciB0aGlzIE9wZW5BSSBjb25maWd1cmF0aW9uLg","name":"Introducing GPT-6.1 Sol","notes":"Reported reasoning effort max. Share of answers with at least one factual error on de-identified conversations where users flagged an earlier model error; deliberately difficult, not representative of typical usage. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration."},{"benchmarkId":"failure-to-disclose-a-broken-search-tool-sol-6-1-launch-safety-stress-test","id":"failure-to-disclose-a-broken-search-tool-sol-6-1-launch-safety-stress-test:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIEFkdmVyc2FyaWFsIHRlc3Qgb2Ygd2hldGhlciBhZ2VudHMgZGlzY2xvc2UgYSBicm9rZW4gc2VhcmNoIHRvb2wuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","name":"Introducing GPT-6.1 Sol","notes":"Reported reasoning effort max. Adversarial test of whether agents disclose a broken search tool. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration."},{"benchmarkId":"reviewer-bypass-attempts-sol-6-1-launch-safety-stress-test","id":"reviewer-bypass-attempts-sol-6-1-launch-safety-stress-test:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIEF0dGVtcHRzIHRvIHdvcmsgYXJvdW5kIGFuIGF1dG9tYXRlZCBzYWZldHkgcmV2aWV3ZXIgYmxvY2tpbmcgYW4gYWN0aW9uLiBPcGVuQUkgcmVzZWFyY2ggZW52aXJvbm1lbnQgb3IgQVBJOyBzYW1lIG5hbWVkIG1ldHJpYyBhbmQgZXZhbHVhdGlvbiBjaGFydC4gTm8gZmFsbGJhY2sgaXMgcmVwb3J0ZWQgZm9yIHRoaXMgT3BlbkFJIGNvbmZpZ3VyYXRpb24u","name":"Introducing GPT-6.1 Sol","notes":"Reported reasoning effort max. Attempts to work around an automated safety reviewer blocking an action. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration."},{"benchmarkId":"warning-circumvention-sol-6-1-launch-safety-stress-test","id":"warning-circumvention-sol-6-1-launch-safety-stress-test:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIFJlc3RyaWN0aW9uLWNpcmN1bXZlbnRpb24gYXR0ZW1wdHMgaW4gZGVsaWJlcmF0ZWx5IGRpZmZpY3VsdCBsb3ctc3Rha2VzIGNhc2VzIHdpdGhvdXQgZnVsbCBwcm9kdWN0IHNhZmVndWFyZHMuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","name":"Introducing GPT-6.1 Sol","notes":"Reported reasoning effort max. Restriction-circumvention attempts in deliberately difficult low-stakes cases without full product safeguards. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration."},{"benchmarkId":"computer-use-safety-stress-test-sol-6-1-launch-safety-stress-test","id":"computer-use-safety-stress-test-sol-6-1-launch-safety-stress-test:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCB4aGlnaC4gVW5pbnRlbmRlZCBvdXRjb21lcyBpbiBkZWxpYmVyYXRlbHkgYWR2ZXJzYXJpYWwgY29tcHV0ZXItIGFuZCBicm93c2VyLXVzZSB3b3JrcGxhY2UgdGFza3M7IHRoZSB1cGRhdGVkIGhhcmRlciBzYWZldHkgc3Vic2V0LCBub3QgT1NXb3JsZCB0YXNrIHN1Y2Nlc3MuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","name":"Introducing GPT-6.1 Sol","notes":"Reported reasoning effort xhigh. Unintended outcomes in deliberately adversarial computer- and browser-use workplace tasks; the updated harder safety subset, not OSWorld task success. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration."}],"ingestRuns":[{"adapter":"arc-agi-2","finishedAt":"2026-09-25T00:00:44.349Z","id":"ingest-arc-agi-2-402780032ab1","note":"Official ARC Prize ARC-AGI-2 board (https://arcprize.org/leaderboard, arc-agi-2-official, v2_Semi_Private). 51 mapped. hash=402780032ab1b4c564613360672735d148299476450e74eb265389007eb6292e","ok":true,"startedAt":"2026-09-25T00:00:44.349Z"},{"adapter":"arc-agi-2","finishedAt":"2026-09-23T00:54:18.197Z","id":"ingest-arc-agi-2-4e77c3315e06","note":"Official ARC Prize ARC-AGI-2 board (https://arcprize.org/leaderboard, arc-agi-2-official, v2_Semi_Private). 50 mapped. hash=4e77c3315e06105dbdf8e87e90d3777fb5a533d6fb5177318c33b484e7c73a26","ok":true,"startedAt":"2026-09-23T00:54:18.197Z"},{"adapter":"arc-agi-2","finishedAt":"2026-09-28T21:43:19.415Z","id":"ingest-arc-agi-2-79030dfe89f1","note":"Official ARC Prize ARC-AGI-2 board (https://arcprize.org/leaderboard, arc-agi-2-official, v2_Semi_Private). 52 mapped. hash=79030dfe89f1ee93b73b3c73a0c583b1a6cd703f3ba34db6eef7fd1580ddc929","ok":true,"startedAt":"2026-09-28T21:43:19.415Z"},{"adapter":"arc-agi-2","finishedAt":"2026-09-29T20:23:07.479Z","id":"ingest-arc-agi-2-ad8fb82d3c8d","note":"Official ARC Prize ARC-AGI-2 board (https://arcprize.org/leaderboard, arc-agi-2-official, v2_Semi_Private). 53 mapped. hash=ad8fb82d3c8d3c6bf0d6db10613c7b9fb882a5278c1fbf5833ee140d01c5e38e","ok":true,"startedAt":"2026-09-29T20:23:07.479Z"},{"adapter":"arc-agi-2","finishedAt":"2026-09-21T18:18:29.075Z","id":"ingest-arc-agi-2-c26215536848","note":"Official ARC Prize ARC-AGI-2 board (https://arcprize.org/leaderboard, arc-agi-2-official, v2_Semi_Private). 48 mapped. hash=c2621553684830980fabd967dbf06ca433fa21bf2d407e2ab3f54f23afdf98aa","ok":true,"startedAt":"2026-09-21T18:18:29.075Z"},{"adapter":"arc-agi-2","finishedAt":"2026-09-11T12:08:13.295Z","id":"ingest-arc-agi-2-efebf61be277","note":"Official ARC Prize ARC-AGI-2 board (https://arcprize.org/leaderboard, arc-agi-2-official, v2_Semi_Private). 48 mapped. hash=efebf61be27786f10137c9a6a133e5007efbf6d397806191a70217c82a6849be","ok":true,"startedAt":"2026-09-11T12:08:13.295Z"},{"adapter":"automationbench-aa","finishedAt":"2026-09-24T00:01:08.699Z","id":"ingest-automationbench-aa-00f1d08c9b0a","note":"Official AA AutomationBench-AA score. Share of objectives completed with no guardrail violation. 48 mapped. hash=00f1d08c9b0a5870cbdaa1b4760cee2f58c8ab51ed9895cb1ad2af7b710816ce","ok":true,"startedAt":"2026-09-24T00:01:08.699Z"},{"adapter":"automationbench-aa","finishedAt":"2026-09-25T00:00:34.700Z","id":"ingest-automationbench-aa-0c9b9b719d27","note":"Official AA AutomationBench-AA score. Share of objectives completed with no guardrail violation. 48 mapped. hash=0c9b9b719d274e3d8b0fc877138381e7499e72f22894fc9bc7b3c4bee3d1f358","ok":true,"startedAt":"2026-09-25T00:00:34.700Z"},{"adapter":"automationbench-aa","finishedAt":"2026-09-23T00:54:03.142Z","id":"ingest-automationbench-aa-3b0a318dba6f","note":"Official AA AutomationBench-AA score. Share of objectives completed with no guardrail violation. 49 mapped. hash=3b0a318dba6f0de25d6fa36b71a0839c36f512c1da8fd3daef6b199a148a1841","ok":true,"startedAt":"2026-09-23T00:54:03.142Z"},{"adapter":"automationbench-aa","finishedAt":"2026-09-28T21:43:07.776Z","id":"ingest-automationbench-aa-4672b1f6b6e3","note":"Official AA AutomationBench-AA score. Share of objectives completed with no guardrail violation. 49 mapped. hash=4672b1f6b6e34d9782a7a88e9c856164e7c3a336051615787cf6cd112a4bacad","ok":true,"startedAt":"2026-09-28T21:43:07.776Z"},{"adapter":"automationbench-aa","finishedAt":"2026-09-23T06:43:03.856Z","id":"ingest-automationbench-aa-529d21dd37e5","note":"Official AA AutomationBench-AA score. Share of objectives completed with no guardrail violation. 48 mapped. hash=529d21dd37e5059faed1364ac2e5f032986482d42bde3c00804d22b1255d03eb","ok":true,"startedAt":"2026-09-23T06:43:03.856Z"},{"adapter":"automationbench-aa","finishedAt":"2026-09-24T06:53:41.663Z","id":"ingest-automationbench-aa-67e1eee44927","note":"Official AA AutomationBench-AA score. Share of objectives completed with no guardrail violation. 48 mapped. hash=67e1eee4492783a2cc61e10a2ca4b77d0b5b9bdc25cbfc75fc9d91d58938ce36","ok":true,"startedAt":"2026-09-24T06:53:41.663Z"},{"adapter":"automationbench-aa","finishedAt":"2026-09-29T20:22:55.777Z","id":"ingest-automationbench-aa-712aa4217411","note":"Official AA AutomationBench-AA score. Share of objectives completed with no guardrail violation. 49 mapped. hash=712aa421741169313d1deaedce75a8ed04d9eede62542d7613b0e0d3c9e05e6b","ok":true,"startedAt":"2026-09-29T20:22:55.777Z"},{"adapter":"automationbench-aa","finishedAt":"2026-09-28T22:13:45.298Z","id":"ingest-automationbench-aa-86900e9188d4","note":"Official AA AutomationBench-AA score. Share of objectives completed with no guardrail violation. 49 mapped. hash=86900e9188d4e59a25b6ae2d83d0e8ea8103f08a5b552228ceb6272f56db27b8","ok":true,"startedAt":"2026-09-28T22:13:45.298Z"},{"adapter":"automationbench-aa","finishedAt":"2026-09-28T00:01:32.231Z","id":"ingest-automationbench-aa-d3333940a5c5","note":"Official AA AutomationBench-AA score. Share of objectives completed with no guardrail violation. 48 mapped. hash=d3333940a5c5ac2bd4674b44b735b1e90db1899929ee63d213af189575565d1f","ok":true,"startedAt":"2026-09-28T00:01:32.231Z"},{"adapter":"autoresearchexam","finishedAt":"2026-09-28T21:43:32.756Z","id":"ingest-autoresearchexam-3d089595005a-8974d7ca","note":"Bespoke Labs AutoResearchExam hidden-test AUARC. 29 tasks, Terminus 2, 24-hour window. Display only. Excluded from weighted Overall.","ok":true,"startedAt":"2026-09-28T21:43:32.756Z"},{"adapter":"autoresearchexam","finishedAt":"2026-09-23T00:54:38.196Z","id":"ingest-autoresearchexam-3d089595005a-f55dfe2b","note":"Bespoke Labs AutoResearchExam hidden-test AUARC. 29 tasks, Terminus 2, 24-hour window. Display only. Excluded from weighted Overall.","ok":true,"startedAt":"2026-09-23T00:54:38.196Z"},{"adapter":"autoresearchexam","finishedAt":"2026-09-29T20:23:22.365Z","id":"ingest-autoresearchexam-3d089595005a-fb878f2f","note":"Bespoke Labs AutoResearchExam hidden-test AUARC. 29 tasks, Terminus 2, 24-hour window. Display only. Excluded from weighted Overall.","ok":true,"startedAt":"2026-09-29T20:23:22.365Z"},{"adapter":"bughunt-bench","finishedAt":"2026-09-28T22:42:32.487Z","id":"ingest-bughunt-bench-28d54cb375cd-8974d7ca","note":"Bug Hunt Bench featured non-superseded rows. Score is (fixed / 105) * 100, planted bugs fixed only. Display only pending comparability review. Not in the weighted ranking.","ok":true,"startedAt":"2026-09-28T22:42:32.487Z"},{"adapter":"bughunt-bench","finishedAt":"2026-09-23T09:17:40.493Z","id":"ingest-bughunt-bench-3b0ee33cbf7f-f55dfe2b","note":"Bug Hunt Bench featured non-superseded rows. Score is (fixed / 105) * 100, planted bugs fixed only. Display only pending comparability review. Not in the weighted ranking.","ok":true,"startedAt":"2026-09-23T09:17:40.493Z"},{"adapter":"bughunt-bench","finishedAt":"2026-09-23T00:54:41.480Z","id":"ingest-bughunt-bench-770e162acc67-f55dfe2b","note":"Bug Hunt Bench featured non-superseded rows. Score is (fixed / 105) * 100, planted bugs fixed only. Display only pending comparability review. Not in the weighted ranking.","ok":true,"startedAt":"2026-09-23T00:54:41.480Z"},{"adapter":"bughunt-bench","finishedAt":"2026-09-24T00:54:54.879Z","id":"ingest-bughunt-bench-8578613b5a3b-f55dfe2b","note":"Bug Hunt Bench featured non-superseded rows. Score is (fixed / 105) * 100, planted bugs fixed only. Display only pending comparability review. Not in the weighted ranking.","ok":true,"startedAt":"2026-09-24T00:54:54.879Z"},{"adapter":"bughunt-bench","finishedAt":"2026-09-24T00:01:31.329Z","id":"ingest-bughunt-bench-86b16c2f2b16-f55dfe2b","note":"Bug Hunt Bench featured non-superseded rows. Score is (fixed / 105) * 100, planted bugs fixed only. Display only pending comparability review. Not in the weighted ranking.","ok":true,"startedAt":"2026-09-24T00:01:31.329Z"},{"adapter":"bughunt-bench","finishedAt":"2026-09-29T20:23:24.438Z","id":"ingest-bughunt-bench-ce001799cfba-fb878f2f","note":"Bug Hunt Bench featured non-superseded rows. Score is (fixed / 105) * 100, planted bugs fixed only. Display only pending comparability review. Not in the weighted ranking.","ok":true,"startedAt":"2026-09-29T20:23:24.438Z"},{"adapter":"bughunt-bench","finishedAt":"2026-09-28T21:43:34.635Z","id":"ingest-bughunt-bench-d58f9add7926-8974d7ca","note":"Bug Hunt Bench featured non-superseded rows. Score is (fixed / 105) * 100, planted bugs fixed only. Display only pending comparability review. Not in the weighted ranking.","ok":true,"startedAt":"2026-09-28T21:43:34.635Z"},{"adapter":"bughunt-bench","finishedAt":"2026-09-24T06:53:59.036Z","id":"ingest-bughunt-bench-d58f9add7926-f55dfe2b","note":"Bug Hunt Bench featured non-superseded rows. Score is (fixed / 105) * 100, planted bugs fixed only. Display only pending comparability review. Not in the weighted ranking.","ok":true,"startedAt":"2026-09-24T06:53:59.036Z"},{"adapter":"bughunt-bench","finishedAt":"2026-09-30T00:22:17.343Z","id":"ingest-bughunt-bench-e184b9c0609a-fb878f2f","note":"Bug Hunt Bench featured non-superseded rows. Score is (fixed / 105) * 100, planted bugs fixed only. Display only pending comparability review. Not in the weighted ranking.","ok":true,"startedAt":"2026-09-30T00:22:17.343Z"},{"adapter":"bughunt-bench","finishedAt":"2026-09-29T22:38:03.400Z","id":"ingest-bughunt-bench-e244f7b0ca68-fb878f2f","note":"Bug Hunt Bench featured non-superseded rows. Score is (fixed / 105) * 100, planted bugs fixed only. Display only pending comparability review. Not in the weighted ranking.","ok":true,"startedAt":"2026-09-29T22:38:03.400Z"},{"adapter":"bughunt-bench","finishedAt":"2026-09-23T06:43:25.691Z","id":"ingest-bughunt-bench-f737b064060e-f55dfe2b","note":"Bug Hunt Bench featured non-superseded rows. Score is (fixed / 105) * 100, planted bugs fixed only. Display only pending comparability review. Not in the weighted ranking.","ok":true,"startedAt":"2026-09-23T06:43:25.691Z"},{"adapter":"bughunt-bench","finishedAt":"2026-09-29T19:46:47.235Z","id":"ingest-bughunt-bench-fd51e329ceb9-8974d7ca","note":"Bug Hunt Bench featured non-superseded rows. Score is (fixed / 105) * 100, planted bugs fixed only. Display only pending comparability review. Not in the weighted ranking.","ok":true,"startedAt":"2026-09-29T19:46:47.235Z"},{"adapter":"deepswe-v1.1","finishedAt":"2026-09-29T20:23:13.220Z","id":"ingest-deepswe-v1.1-eb88d1c756c0","note":"Official DeepSWE v1.1 pass@1 board (https://deepswe.datacurve.ai/, deepswe-v1.1-reported, mini-swe-agent, pass@1, 113 tasks). 28 mapped. hash=eb88d1c756c0ea2ad7ff534dbf9c4ebd36ed73e4ca1b560533ba8ae447f75ee4","ok":true,"startedAt":"2026-09-29T20:23:13.220Z"},{"adapter":"fixture","finishedAt":"2026-09-04T00:00:00Z","id":"ingest-fixture-v0","note":"Checked-in v0 fixture. Not a live adapter.","ok":true,"startedAt":"2026-09-04T00:00:00Z"},{"adapter":"gdpval-aa","finishedAt":"2026-09-24T06:53:39.035Z","id":"ingest-gdpval-aa-07dd3696c36c","note":"Official AA GDPval-AA v2 board. 37 mapped. hash=07dd3696c36ca7a6efb752c6e2a329077420655d07f0b9a5d4781b76219df9ec","ok":true,"startedAt":"2026-09-24T06:53:39.035Z"},{"adapter":"gdpval-aa","finishedAt":"2026-09-24T00:01:05.477Z","id":"ingest-gdpval-aa-09a550a9c8fd","note":"Official AA GDPval-AA v2 board. 37 mapped. hash=09a550a9c8fd2864968643b0efbb82262e5445e14dfd4758641aa6855d0828d5","ok":true,"startedAt":"2026-09-24T00:01:05.477Z"},{"adapter":"gdpval-aa","finishedAt":"2026-09-12T08:02:31.496Z","id":"ingest-gdpval-aa-0a9ee574f268","note":"Official AA GDPval-AA v2 board. 41 mapped. hash=0a9ee574f2681ec9d8ac46138442e68a2107c97f934d43ba6a19a3009e670a6f","ok":true,"startedAt":"2026-09-12T08:02:31.496Z"},{"adapter":"gdpval-aa","finishedAt":"2026-09-30T00:21:57.559Z","id":"ingest-gdpval-aa-0efcc4622a07","note":"Official AA GDPval-AA v2 board. 37 mapped. hash=0efcc4622a079db511b05929d60904491e7f3a9fe94fa2f6df2546751ac1df08","ok":true,"startedAt":"2026-09-30T00:21:57.559Z"},{"adapter":"gdpval-aa","finishedAt":"2026-09-29T20:22:53.438Z","id":"ingest-gdpval-aa-0f05423a21e2","note":"Official AA GDPval-AA v2 board. 37 mapped. hash=0f05423a21e29cd0fffaa053b4e7801e2821024b0f34b55d036152867e121499","ok":true,"startedAt":"2026-09-29T20:22:53.438Z"},{"adapter":"gdpval-aa","finishedAt":"2026-09-09T06:00:28.240Z","id":"ingest-gdpval-aa-1327d22f7245","note":"Official AA GDPval-AA v2 board. 31 mapped. hash=1327d22f7245c5588f78bed7c8af86c62094b629a04b186bc617b2fcb0e60fbf","ok":true,"startedAt":"2026-09-09T06:00:28.240Z"},{"adapter":"gdpval-aa","finishedAt":"2026-09-15T06:35:54.839Z","id":"ingest-gdpval-aa-150695b1ae15","note":"Official AA GDPval-AA v2 board. 41 mapped. hash=150695b1ae15c720ef8a19c064a95896fd070f3c0d5ccd3ca56a7b77ea321b69","ok":true,"startedAt":"2026-09-15T06:35:54.839Z"},{"adapter":"gdpval-aa","finishedAt":"2026-09-17T00:28:55.990Z","id":"ingest-gdpval-aa-1b0e2666cd49","note":"Official AA GDPval-AA v2 board. 40 mapped. hash=1b0e2666cd49c7ef8e52dedce10e9bace4346b6b71685fed5ed2bf76d211e7d3","ok":true,"startedAt":"2026-09-17T00:28:55.990Z"},{"adapter":"gdpval-aa","finishedAt":"2026-09-21T18:18:17.227Z","id":"ingest-gdpval-aa-1f53d558f193","note":"Official AA GDPval-AA v2 board. 37 mapped. hash=1f53d558f193b1fadade6eef02d3eaa1ff5044f7ed74369870f1f7f8c57223ad","ok":true,"startedAt":"2026-09-21T18:18:17.227Z"},{"adapter":"gdpval-aa","finishedAt":"2026-09-06T15:46:12.431Z","id":"ingest-gdpval-aa-2a45c724c848","note":"Official AA GDPval-AA v2 board. 31 mapped. hash=2a45c724c8487dbcd2d0bbbadef1c15186889a7f26ccf98c311fc7411e8cafab","ok":true,"startedAt":"2026-09-06T15:46:12.431Z"},{"adapter":"gdpval-aa","finishedAt":"2026-09-11T12:08:01.430Z","id":"ingest-gdpval-aa-33526d0ea86c","note":"Official AA GDPval-AA v2 board. 41 mapped. hash=33526d0ea86c381dec15a78e33d949a410340f049ae4f779aafc1525a99e4470","ok":true,"startedAt":"2026-09-11T12:08:01.430Z"},{"adapter":"gdpval-aa","finishedAt":"2026-09-20T00:00:48.557Z","id":"ingest-gdpval-aa-35d8e4bb317a","note":"Official AA GDPval-AA v2 board. 37 mapped. hash=35d8e4bb317a4f9bce1db925f034b0e6e350d0c0e2a1efb9b8f04563e6fb8894","ok":true,"startedAt":"2026-09-20T00:00:48.557Z"},{"adapter":"gdpval-aa","finishedAt":"2026-09-13T00:00:54.254Z","id":"ingest-gdpval-aa-471b6c4a089e","note":"Official AA GDPval-AA v2 board. 41 mapped. hash=471b6c4a089ee3f884920da7644ff703bd899585d67725554666b0781fd51ded","ok":true,"startedAt":"2026-09-13T00:00:54.254Z"},{"adapter":"gdpval-aa","finishedAt":"2026-09-25T00:00:31.495Z","id":"ingest-gdpval-aa-508c6ebc7952","note":"Official AA GDPval-AA v2 board. 37 mapped. hash=508c6ebc7952803a40974e4ab6e1b7a1b95f6dc841f9b82cdb9a6e678b2cc31d","ok":true,"startedAt":"2026-09-25T00:00:31.495Z"},{"adapter":"gdpval-aa","finishedAt":"2026-09-12T08:38:19.822Z","id":"ingest-gdpval-aa-58c478670eac","note":"Official AA GDPval-AA v2 board. 41 mapped. hash=58c478670eac759d50580d256fac007239d2347dd7ce881834d11d7b518aa4df","ok":true,"startedAt":"2026-09-12T08:38:19.822Z"},{"adapter":"gdpval-aa","finishedAt":"2026-09-15T00:00:47.232Z","id":"ingest-gdpval-aa-6167623c6cde","note":"Official AA GDPval-AA v2 board. 41 mapped. hash=6167623c6cde240fbf3e94e4cefe9f8abb3aa3c820858cc82125d46962f3afd0","ok":true,"startedAt":"2026-09-15T00:00:47.232Z"},{"adapter":"gdpval-aa","finishedAt":"2026-09-17T00:00:49.790Z","id":"ingest-gdpval-aa-6230760ba277","note":"Official AA GDPval-AA v2 board. 40 mapped. hash=6230760ba2779b8fbbc2f6ea48d2a10528ac94eeacc4d5e1887c060d3c2a627c","ok":true,"startedAt":"2026-09-17T00:00:49.790Z"},{"adapter":"gdpval-aa","finishedAt":"2026-09-27T00:01:54.206Z","id":"ingest-gdpval-aa-69430119220e","note":"Official AA GDPval-AA v2 board. 37 mapped. hash=69430119220e12e4a614687410afbdc1fa663c5165c56e138f0679550f9a4cbe","ok":true,"startedAt":"2026-09-27T00:01:54.206Z"},{"adapter":"gdpval-aa","finishedAt":"2026-09-10T06:00:21.353Z","id":"ingest-gdpval-aa-6ba4006169f8","note":"Official AA GDPval-AA v2 board. 31 mapped. hash=6ba4006169f8086956ab4b3733c867af002211ed4085b2812f87ec8c8e90fc8e","ok":true,"startedAt":"2026-09-10T06:00:21.353Z"},{"adapter":"gdpval-aa","finishedAt":"2026-09-22T00:00:45.814Z","id":"ingest-gdpval-aa-6f6ecc641d0c","note":"Official AA GDPval-AA v2 board. 37 mapped. hash=6f6ecc641d0cfc7593355ae0242766970bc55c5b4f4d5262a2a13a7aa427f831","ok":true,"startedAt":"2026-09-22T00:00:45.814Z"},{"adapter":"gdpval-aa","finishedAt":"2026-09-19T07:56:00.690Z","id":"ingest-gdpval-aa-7d1d4cbf4497","note":"Official AA GDPval-AA v2 board. 40 mapped. hash=7d1d4cbf44978941f5507b0a4a39e7ff28cd562698c1169505863ec7d3f401e6","ok":true,"startedAt":"2026-09-19T07:56:00.690Z"},{"adapter":"gdpval-aa","finishedAt":"2026-09-14T00:01:02.780Z","id":"ingest-gdpval-aa-8f890d2496df","note":"Official AA GDPval-AA v2 board. 41 mapped. hash=8f890d2496df22e9455d71478f77e82ec12cd3ec185d5c0855df9c101590a8e2","ok":true,"startedAt":"2026-09-14T00:01:02.780Z"},{"adapter":"gdpval-aa","finishedAt":"2026-09-28T00:01:30.071Z","id":"ingest-gdpval-aa-924ed68cfb6e","note":"Official AA GDPval-AA v2 board. 37 mapped. hash=924ed68cfb6ebb017222da0ea6413b05bb9e1dad790c6037eb557e57b48218ff","ok":true,"startedAt":"2026-09-28T00:01:30.071Z"},{"adapter":"gdpval-aa","finishedAt":"2026-09-23T00:54:00.009Z","id":"ingest-gdpval-aa-a5696440b5c7","note":"Official AA GDPval-AA v2 board. 37 mapped. hash=a5696440b5c7899212065f510fe906b58ecc9d5e4c872ea5ad3d948cfaf5aef6","ok":true,"startedAt":"2026-09-23T00:54:00.009Z"},{"adapter":"gdpval-aa","finishedAt":"2026-09-16T00:01:13.451Z","id":"ingest-gdpval-aa-a73a52b0d734","note":"Official AA GDPval-AA v2 board. 41 mapped. hash=a73a52b0d734c9de59dc6eaef4f26a6d0411a47077ac3a603b3de34880d4833e","ok":true,"startedAt":"2026-09-16T00:01:13.451Z"},{"adapter":"gdpval-aa","finishedAt":"2026-09-29T23:10:58.031Z","id":"ingest-gdpval-aa-c084d063b295","note":"Official AA GDPval-AA v2 board. 37 mapped. hash=c084d063b2953813e78d748b4e75d8588b33dbdeb1d8ed79d4edb9e5ff88a418","ok":true,"startedAt":"2026-09-29T23:10:58.031Z"},{"adapter":"gdpval-aa","finishedAt":"2026-09-08T06:00:48.428Z","id":"ingest-gdpval-aa-cc112b57ecc6","note":"Official AA GDPval-AA v2 board. 31 mapped. hash=cc112b57ecc6e68f21dd3c4859d1210228390cbd766ce39bad758ee6e521c233","ok":true,"startedAt":"2026-09-08T06:00:48.428Z"},{"adapter":"gdpval-aa","finishedAt":"2026-09-07T06:00:07.822Z","id":"ingest-gdpval-aa-e73ea462e58c","note":"Official AA GDPval-AA v2 board. 31 mapped. hash=e73ea462e58c14cff71ab3cd1f053f9222e17a7db89adca89307035f070eb696","ok":true,"startedAt":"2026-09-07T06:00:07.822Z"},{"adapter":"gdpval-aa","finishedAt":"2026-09-18T00:00:47.995Z","id":"ingest-gdpval-aa-e8ab089121f2","note":"Official AA GDPval-AA v2 board. 40 mapped. hash=e8ab089121f25c7b1028f1a237eb0fed844b8b026937b2e2c8d6836c84195c55","ok":true,"startedAt":"2026-09-18T00:00:47.995Z"},{"adapter":"gdpval-aa","finishedAt":"2026-09-21T00:00:55.762Z","id":"ingest-gdpval-aa-ec955c5d8d25","note":"Official AA GDPval-AA v2 board. 37 mapped. hash=ec955c5d8d255b5584b0c2e14b6770a84751dea841ebc802017e2b721aa96b71","ok":true,"startedAt":"2026-09-21T00:00:55.762Z"},{"adapter":"gdpval-aa","finishedAt":"2026-09-28T21:43:05.769Z","id":"ingest-gdpval-aa-f7239eb6b9aa","note":"Official AA GDPval-AA v2 board. 37 mapped. hash=f7239eb6b9aa80346dc60bbd0d600826e1ae123d78186a1f38ce071dbe21b5ff","ok":true,"startedAt":"2026-09-28T21:43:05.769Z"},{"adapter":"gdpval-aa","finishedAt":"2026-09-12T00:00:42.891Z","id":"ingest-gdpval-aa-fbab5a677c48","note":"Official AA GDPval-AA v2 board. 41 mapped. hash=fbab5a677c4819567541930f35be727e62e4c506718367f8bc37114a7685fdaa","ok":true,"startedAt":"2026-09-12T00:00:42.891Z"},{"adapter":"gpqa-diamond","finishedAt":"2026-09-29T20:23:15.713Z","id":"ingest-gpqa-diamond-14ee8600a9d3","note":"Artificial Analysis GPQA Diamond independent run (https://artificialanalysis.ai/evaluations/gpqa-diamond, gpqa-diamond-reported, n=198). Paper repo https://github.com/idavidrein/gpqa has no live official board. 18 mapped. hash=14ee8600a9d371c0a5471c3bfeb3d2a5eee0d2c4b90037a8a2cb9392fceadffc","ok":true,"startedAt":"2026-09-29T20:23:15.713Z"},{"adapter":"gpqa-diamond","finishedAt":"2026-09-17T00:01:01.190Z","id":"ingest-gpqa-diamond-39516434320c","note":"Artificial Analysis GPQA Diamond independent run (https://artificialanalysis.ai/evaluations/gpqa-diamond, gpqa-diamond-reported, n=198). Paper repo https://github.com/idavidrein/gpqa has no live official board. 23 mapped. hash=39516434320c9f5f06fbbe5e0c8251ad3be5229b56e52c64236fab192a532790","ok":true,"startedAt":"2026-09-17T00:01:01.190Z"},{"adapter":"gpqa-diamond","finishedAt":"2026-09-11T12:08:23.704Z","id":"ingest-gpqa-diamond-41e862296ea6","note":"Artificial Analysis GPQA Diamond independent run (https://artificialanalysis.ai/evaluations/gpqa-diamond, gpqa-diamond-reported, n=198). Paper repo https://github.com/idavidrein/gpqa has no live official board. 23 mapped. hash=41e862296ea6ad5c878f2fc2162f0f32cebc80d9190b47f40654aff8f400be4b","ok":true,"startedAt":"2026-09-11T12:08:23.704Z"},{"adapter":"gpqa-diamond","finishedAt":"2026-09-25T00:29:21.379Z","id":"ingest-gpqa-diamond-4dda8dbfc4a0","note":"Artificial Analysis GPQA Diamond independent run (https://artificialanalysis.ai/evaluations/gpqa-diamond, gpqa-diamond-reported, n=198). Paper repo https://github.com/idavidrein/gpqa has no live official board. 19 mapped. hash=4dda8dbfc4a017eb84ade8544e21bda55616c5f75fbe9e54db5471f0f885ff31","ok":true,"startedAt":"2026-09-25T00:29:21.379Z"},{"adapter":"gpqa-diamond","finishedAt":"2026-09-23T00:54:29.302Z","id":"ingest-gpqa-diamond-7ed1056a8fa5","note":"Artificial Analysis GPQA Diamond independent run (https://artificialanalysis.ai/evaluations/gpqa-diamond, gpqa-diamond-reported, n=198). Paper repo https://github.com/idavidrein/gpqa has no live official board. 19 mapped. hash=7ed1056a8fa5c7e1f8a70c6c8af32229bde7dd58fd1c96b65cbef17f96bc9cf0","ok":true,"startedAt":"2026-09-23T00:54:29.302Z"},{"adapter":"gpqa-diamond","finishedAt":"2026-09-21T18:18:41.497Z","id":"ingest-gpqa-diamond-8509afacdd47","note":"Artificial Analysis GPQA Diamond independent run (https://artificialanalysis.ai/evaluations/gpqa-diamond, gpqa-diamond-reported, n=198). Paper repo https://github.com/idavidrein/gpqa has no live official board. 22 mapped. hash=8509afacdd470afd77999c465efce0cb89ba00c09012d8c7ddf9640519f88a60","ok":true,"startedAt":"2026-09-21T18:18:41.497Z"},{"adapter":"gpqa-diamond","finishedAt":"2026-09-30T00:22:10.291Z","id":"ingest-gpqa-diamond-a84c8109bd68","note":"Artificial Analysis GPQA Diamond independent run (https://artificialanalysis.ai/evaluations/gpqa-diamond, gpqa-diamond-reported, n=198). Paper repo https://github.com/idavidrein/gpqa has no live official board. 16 mapped. hash=a84c8109bd682ef82be5386d44eaf3730cd9adfb325bd0eefdf9ca8a40d00018","ok":true,"startedAt":"2026-09-30T00:22:10.291Z"},{"adapter":"gpqa-diamond","finishedAt":"2026-09-08T06:00:56.707Z","id":"ingest-gpqa-diamond-bea3d6f4c97e","note":"Artificial Analysis GPQA Diamond independent run (https://artificialanalysis.ai/evaluations/gpqa-diamond, gpqa-diamond-reported, n=198). Paper repo https://github.com/idavidrein/gpqa has no live official board. 23 mapped. hash=bea3d6f4c97e748d4049a5e9397ee9d1136728232f2fd4e3d54ce2b77ef3c870","ok":true,"startedAt":"2026-09-08T06:00:56.707Z"},{"adapter":"gpqa-diamond","finishedAt":"2026-09-28T21:43:26.476Z","id":"ingest-gpqa-diamond-c2a5b4c38dab","note":"Artificial Analysis GPQA Diamond independent run (https://artificialanalysis.ai/evaluations/gpqa-diamond, gpqa-diamond-reported, n=198). Paper repo https://github.com/idavidrein/gpqa has no live official board. 18 mapped. hash=c2a5b4c38dab91537e858fe5bfcb7840682f2b44badf2d35ee5b745c8da39ddb","ok":true,"startedAt":"2026-09-28T21:43:26.476Z"},{"adapter":"gpqa-diamond","finishedAt":"2026-09-18T00:00:58.832Z","id":"ingest-gpqa-diamond-ca410b4c77b2","note":"Artificial Analysis GPQA Diamond independent run (https://artificialanalysis.ai/evaluations/gpqa-diamond, gpqa-diamond-reported, n=198). Paper repo https://github.com/idavidrein/gpqa has no live official board. 23 mapped. hash=ca410b4c77b22441a64def7ba8a2961e8ea9ad7c4bc77d17669a8a5ba434abd7","ok":true,"startedAt":"2026-09-18T00:00:58.832Z"},{"adapter":"gpqa-diamond","finishedAt":"2026-09-06T15:46:37.137Z","id":"ingest-gpqa-diamond-ccdfc380ea0e","note":"Artificial Analysis GPQA Diamond independent run (https://artificialanalysis.ai/evaluations/gpqa-diamond, gpqa-diamond-reported, n=198). Paper repo https://github.com/idavidrein/gpqa has no live official board. 21 mapped. hash=ccdfc380ea0e9cc846cfbd710cacc7f746b86582cf03f36f5954f3b443fc81a2","ok":true,"startedAt":"2026-09-06T15:46:37.137Z"},{"adapter":"hle","finishedAt":"2026-09-06T15:46:22.229Z","id":"ingest-hle-c5396957877c","note":"Official CAIS/Scale HLE board (https://lastexam.ai/, hle-no-tools). 7 mapped. hash=c5396957877c522ec0f2c480f44774aa8a322a177c3bf33b9f551d8ad4665dee","ok":true,"startedAt":"2026-09-06T15:46:22.229Z"},{"adapter":"hle","finishedAt":"2026-09-29T20:23:05.709Z","id":"ingest-hle-f549f30a8404","note":"Official CAIS/Scale HLE board (https://lastexam.ai/, hle-no-tools). 8 mapped. hash=f549f30a840454e6cd77755f3bd139f7a0e59bc238c1859c75074f78fca7d95d","ok":true,"startedAt":"2026-09-29T20:23:05.709Z"},{"adapter":"livecodebench","finishedAt":"2026-09-06T15:46:31.323Z","id":"ingest-livecodebench-41a717ad7cd3","note":"Official LiveCodeBench pass@1 board (https://livecodebench.github.io/leaderboard.html, livecodebench-reported, pass@1, 2024-08-01..2025-05-01). 2 mapped. hash=41a717ad7cd373660318d183ec998f51eaebc1ed6d0fcbc750002f61911b4704","ok":true,"startedAt":"2026-09-06T15:46:31.323Z"},{"adapter":"livecodebench","finishedAt":"2026-09-29T20:23:11.465Z","id":"ingest-livecodebench-ad3f286332a4","note":"Official LiveCodeBench pass@1 board (https://livecodebench.github.io/leaderboard.html, livecodebench-reported, pass@1, 2024-08-01..2025-05-01). 9 mapped. hash=ad3f286332a4d41c2928816e3636d084676d48b112b5250e90a8c211f0e57530","ok":true,"startedAt":"2026-09-29T20:23:11.465Z"},{"adapter":"lmarena-text","finishedAt":"2026-09-12T08:02:29.416Z","id":"ingest-lmarena-text-465fe8c26c46","note":"Official LMArena Text Arena dump (text_style_control overall). 115 mapped. hash=465fe8c26c46508b8e4b2995923110399f65a0b996dbb4fc0b013024ce47d087","ok":true,"startedAt":"2026-09-12T08:02:29.416Z"},{"adapter":"lmarena-text","finishedAt":"2026-09-11T12:07:58.579Z","id":"ingest-lmarena-text-8070dca8eca7","note":"Official LMArena Text Arena dump (text_style_control overall). 114 mapped. hash=8070dca8eca7b8ee1fda002af9aa2dc8fa92886f4b36458d876c4dc23527947a","ok":true,"startedAt":"2026-09-11T12:07:58.579Z"},{"adapter":"lmarena-text","finishedAt":"2026-09-29T20:22:50.879Z","id":"ingest-lmarena-text-c660aff40a87","note":"Official LMArena Text Arena dump (text_style_control overall). 114 mapped. hash=c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","ok":true,"startedAt":"2026-09-29T20:22:50.879Z"},{"adapter":"lmarena-text","finishedAt":"2026-09-23T00:53:56.522Z","id":"ingest-lmarena-text-f75684a872fa","note":"Official LMArena Text Arena dump (text_style_control overall). 115 mapped. hash=f75684a872fae40e83c199ec813ef2029abb1478bc4064395524a9616b9e2a28","ok":true,"startedAt":"2026-09-23T00:53:56.522Z"},{"adapter":"mmlu-pro","finishedAt":"2026-09-29T20:23:17.528Z","id":"ingest-mmlu-pro-df03094a9acf","note":"TIGER-Lab MMLU-Pro official board (https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro, dump https://huggingface.co/datasets/TIGER-Lab/mmlu_pro_leaderboard_submission/resolve/main/results.csv, mmlu-pro-reported, n=12032). Paper repo https://github.com/TIGER-AI-Lab/MMLU-Pro has no in-repo live board; the HF space CSV is the official dump. 52 mapped. hash=df03094a9acfd3c1a1bd14b2b3fdb877fbd3e53a121c77235cb5895947548703","ok":true,"startedAt":"2026-09-29T20:23:17.528Z"},{"adapter":"osworld-verified","finishedAt":"2026-09-29T20:23:09.533Z","id":"ingest-osworld-verified-42c386263956","note":"Official OSWorld-Verified board (https://os-world.github.io/, osworld-verified-reported, General model). 6 mapped. hash=42c3862639569a2c686cd8665e632fd4e22ba61e0f565d97cdc01fb786f9fe1e","ok":true,"startedAt":"2026-09-29T20:23:09.533Z"},{"adapter":"real-swe","finishedAt":"2026-09-29T19:46:48.949Z","id":"ingest-real-swe-330eaaa5040c-8974d7ca","note":"September 2026 public board. Eight model+harness rows, 10 private tasks, 8 runs, high reasoning, native harnesses. Display only. Excluded from weighted Overall and capability weights. Public HTML only; /api/ is disallowed.","ok":true,"startedAt":"2026-09-29T19:46:48.949Z"},{"adapter":"real-swe","finishedAt":"2026-09-29T20:23:27.185Z","id":"ingest-real-swe-330eaaa5040c-fb878f2f","note":"September 2026 public board. Eight model+harness rows, 10 private tasks, 8 runs, high reasoning, native harnesses. Display only. Excluded from weighted Overall and capability weights. Public HTML only; /api/ is disallowed.","ok":true,"startedAt":"2026-09-29T20:23:27.185Z"},{"adapter":"real-swe","finishedAt":"2026-09-28T21:43:36.241Z","id":"ingest-real-swe-5afe8e567435-8974d7ca","note":"September 2026 public board. Eight model+harness rows, 10 private tasks, 8 runs, high reasoning, native harnesses. Display only. Excluded from weighted Overall and capability weights. Public HTML only; /api/ is disallowed.","ok":true,"startedAt":"2026-09-28T21:43:36.241Z"},{"adapter":"real-swe","finishedAt":"2026-09-24T06:54:01.623Z","id":"ingest-real-swe-5afe8e567435-f55dfe2b","note":"September 2026 public board. Eight model+harness rows, 10 private tasks, 8 runs, high reasoning, native harnesses. Display only. Excluded from weighted Overall and capability weights. Public HTML only; /api/ is disallowed.","ok":true,"startedAt":"2026-09-24T06:54:01.623Z"},{"adapter":"real-swe","finishedAt":"2026-09-24T00:01:33.867Z","id":"ingest-real-swe-89dc6e043fb3-f55dfe2b","note":"September 2026 public board. Eight model+harness rows, 10 private tasks, 8 runs, high reasoning, native harnesses. Display only. Excluded from weighted Overall and capability weights. Public HTML only; /api/ is disallowed.","ok":true,"startedAt":"2026-09-24T00:01:33.867Z"},{"adapter":"real-swe","finishedAt":"2026-09-23T09:17:43.213Z","id":"ingest-real-swe-e3caa9c8e086-f55dfe2b","note":"September 2026 public board. Eight model+harness rows, 10 private tasks, 8 runs, high reasoning, native harnesses. Display only. Excluded from weighted Overall and capability weights. Public HTML only; /api/ is disallowed.","ok":true,"startedAt":"2026-09-23T09:17:43.213Z"},{"adapter":"reviewed-first-party","finishedAt":"2026-09-06T00:00:00Z","id":"ingest-reported-2026-09-06","note":"Source dates and review decisions are attached to individual observations.","ok":true,"startedAt":"2026-09-06T00:00:00Z"},{"adapter":"swe-bench-pro","finishedAt":"2026-09-29T20:23:03.845Z","id":"ingest-swe-bench-pro-6112d01393e8","note":"Official SWE-bench Pro public board (mini-swe-agent). 4 mapped. hash=6112d01393e8e3b9a1e9c78a6f2a4bf10d22490152d785812020f0f35ade034a","ok":true,"startedAt":"2026-09-29T20:23:03.845Z"},{"adapter":"swe-bench-verified","finishedAt":"2026-09-29T20:22:38.760Z","id":"ingest-swe-verified-077d8c0cfcca","note":"Official SWE-bench Verified dump. 19 mapped. hash=077d8c0cfcca80400745914180b81849577184d691fd8ed35f9832d3fe5368aa","ok":true,"startedAt":"2026-09-29T20:22:38.760Z"},{"adapter":"terminal-bench-2.1","finishedAt":"2026-09-29T20:22:59.155Z","id":"ingest-terminal-bench-2.1-a98710a64150","note":"Official Terminal-Bench 2.1 board (Terminus 2). 2 mapped. hash=a98710a64150db56e74858137de32081bc1e20692a2d140544903b525e654d41","ok":true,"startedAt":"2026-09-29T20:22:59.155Z"},{"adapter":"terminal-bench-4.0","finishedAt":"2026-09-29T20:23:01.873Z","id":"ingest-terminal-bench-4.0-8f0c0dd2a57b","note":"Official Terminal-Bench 4.0 board (Harbor terminal-bench/terminal-bench 4-0-0, native agents, best display row per model). 15 mapped. hash=8f0c0dd2a57be01a51043c0bbeefdadb7758503bd4df7a44830d79e17ca20074","ok":true,"startedAt":"2026-09-29T20:23:01.873Z"},{"adapter":"vulcanbench-v3","finishedAt":"2026-09-11T12:08:29.558Z","id":"ingest-vulcanbench-v3-2f58434bbe3d-740f6b18","note":"Frozen VulcanBench v3: 23 tasks. Bare-bones API reports, separate effort configurations. Fable refusal fallback and extended-budget Kimi excluded. Display only; protocols and repeats differ. See source reports for caveats.","ok":true,"startedAt":"2026-09-11T12:08:29.558Z"},{"adapter":"vulcanbench-v3","finishedAt":"2026-09-28T21:43:30.300Z","id":"ingest-vulcanbench-v3-3cbb0234e083-8974d7ca","note":"Frozen VulcanBench v3: 23 tasks. Bare-bones API reports, separate effort configurations. Fable refusal fallback and extended-budget Kimi excluded. Display only; protocols and repeats differ. See source reports for caveats.","ok":true,"startedAt":"2026-09-28T21:43:30.300Z"},{"adapter":"vulcanbench-v3","finishedAt":"2026-09-24T00:01:26.669Z","id":"ingest-vulcanbench-v3-3cbb0234e083-f55dfe2b","note":"Frozen VulcanBench v3: 23 tasks. Bare-bones API reports, separate effort configurations. Fable refusal fallback and extended-budget Kimi excluded. Display only; protocols and repeats differ. See source reports for caveats.","ok":true,"startedAt":"2026-09-24T00:01:26.669Z"},{"adapter":"vulcanbench-v3","finishedAt":"2026-09-14T00:01:15.533Z","id":"ingest-vulcanbench-v3-92383405bcc1-740f6b18","note":"Frozen VulcanBench v3: 23 tasks. Bare-bones API reports, separate effort configurations. Fable refusal fallback and extended-budget Kimi excluded. Display only; protocols and repeats differ. See source reports for caveats.","ok":true,"startedAt":"2026-09-14T00:01:15.533Z"},{"adapter":"vulcanbench-v3","finishedAt":"2026-09-21T18:18:47.812Z","id":"ingest-vulcanbench-v3-92383405bcc1-87c2b6bc","note":"Frozen VulcanBench v3: 23 tasks. Bare-bones API reports, separate effort configurations. Fable refusal fallback and extended-budget Kimi excluded. Display only; protocols and repeats differ. See source reports for caveats.","ok":true,"startedAt":"2026-09-21T18:18:47.812Z"},{"adapter":"vulcanbench-v3","finishedAt":"2026-09-23T00:54:34.891Z","id":"ingest-vulcanbench-v3-92383405bcc1-f55dfe2b","note":"Frozen VulcanBench v3: 23 tasks. Bare-bones API reports, separate effort configurations. Fable refusal fallback and extended-budget Kimi excluded. Display only; protocols and repeats differ. See source reports for caveats.","ok":true,"startedAt":"2026-09-23T00:54:34.891Z"},{"adapter":"vulcanbench-v3","finishedAt":"2026-09-29T19:46:42.727Z","id":"ingest-vulcanbench-v3-da2f9d24acc8-8974d7ca","note":"Frozen VulcanBench v3: 23 tasks. Bare-bones API reports, separate effort configurations. Fable refusal fallback and extended-budget Kimi excluded. Display only; protocols and repeats differ. See source reports for caveats.","ok":true,"startedAt":"2026-09-29T19:46:42.727Z"},{"adapter":"vulcanbench-v3","finishedAt":"2026-09-29T20:23:19.876Z","id":"ingest-vulcanbench-v3-da2f9d24acc8-fb878f2f","note":"Frozen VulcanBench v3: 23 tasks. Bare-bones API reports, separate effort configurations. Fable refusal fallback and extended-budget Kimi excluded. Display only; protocols and repeats differ. See source reports for caveats.","ok":true,"startedAt":"2026-09-29T20:23:19.876Z"},{"adapter":"launch-page-review","finishedAt":"2026-09-06T00:00:00Z","id":"launch-amazon","note":"Amazon Nova 2 technical report. Date records source review, not evaluation.","ok":true,"startedAt":"2026-09-06T00:00:00Z"},{"adapter":"launch-page-review","finishedAt":"2026-09-30T00:00:00Z","id":"launch-anthropic-fable-5-1-card","note":"Claude Fable 5.1 and Claude Mythos 5.1 System Card. Date records source review, not evaluation.","ok":true,"startedAt":"2026-09-30T00:00:00Z"},{"adapter":"launch-page-review","finishedAt":"2026-09-22T00:00:00Z","id":"launch-anthropic-opus-5-5-card","note":"Claude Opus 5.5 System Card. Date records source review, not evaluation.","ok":true,"startedAt":"2026-09-22T00:00:00Z"},{"adapter":"launch-page-review","finishedAt":"2026-09-06T00:00:00Z","id":"launch-cohere","note":"Command A+ launch benchmarks. Date records source review, not evaluation.","ok":true,"startedAt":"2026-09-06T00:00:00Z"},{"adapter":"launch-page-review","finishedAt":"2026-09-30T00:00:00Z","id":"launch-deepseek","note":"deepseek-ai/DeepSeek-V4-Pro-0813. Date records source review, not evaluation.","ok":true,"startedAt":"2026-09-30T00:00:00Z"},{"adapter":"launch-page-review","finishedAt":"2026-09-06T00:00:00Z","id":"launch-google","note":"Gemini 3.8 Flash launch performance. Date records source review, not evaluation.","ok":true,"startedAt":"2026-09-06T00:00:00Z"},{"adapter":"launch-page-review","finishedAt":"2026-09-06T00:00:00Z","id":"launch-google-gemini-3-8-card","note":"Gemini3.8Flash Model Card. Date records source review, not evaluation.","ok":true,"startedAt":"2026-09-06T00:00:00Z"},{"adapter":"launch-page-review","finishedAt":"2026-09-30T00:00:00Z","id":"launch-kimi","note":"moonshotai/Kimi-K3. Date records source review, not evaluation.","ok":true,"startedAt":"2026-09-30T00:00:00Z"},{"adapter":"launch-page-review","finishedAt":"2026-09-30T00:00:00Z","id":"launch-kimi-k26-card","note":"Kimi K2.6 official model card. Date records source review, not evaluation.","ok":true,"startedAt":"2026-09-30T00:00:00Z"},{"adapter":"launch-page-review","finishedAt":"2026-09-06T00:00:00Z","id":"launch-meta","note":"Muse Spark 1.1 evaluation report Figure44. Date records source review, not evaluation.","ok":true,"startedAt":"2026-09-06T00:00:00Z"},{"adapter":"launch-page-review","finishedAt":"2026-09-06T00:00:00Z","id":"launch-meta-muse-1-3-report","note":"Muse Spark1.3 evaluation methodology. Date records source review, not evaluation.","ok":true,"startedAt":"2026-09-06T00:00:00Z"},{"adapter":"launch-page-review","finishedAt":"2026-09-06T00:00:00Z","id":"launch-microsoft","note":"MAI-Thinking-1 technical report. Date records source review, not evaluation.","ok":true,"startedAt":"2026-09-06T00:00:00Z"},{"adapter":"launch-page-review","finishedAt":"2026-09-06T00:00:00Z","id":"launch-minimax","note":"MiniMax M3 model card benchmark figure. Date records source review, not evaluation.","ok":true,"startedAt":"2026-09-06T00:00:00Z"},{"adapter":"launch-page-review","finishedAt":"2026-09-06T00:00:00Z","id":"launch-mistral","note":"Mistral Medium 3.5 model card performance charts. Date records source review, not evaluation.","ok":true,"startedAt":"2026-09-06T00:00:00Z"},{"adapter":"launch-page-review","finishedAt":"2026-09-06T00:00:00Z","id":"launch-nvidia","note":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16. Date records source review, not evaluation.","ok":true,"startedAt":"2026-09-06T00:00:00Z"},{"adapter":"launch-page-review","finishedAt":"2026-09-06T00:00:00Z","id":"launch-openai-astra-launch","note":"GPT-6 Astra: A new generation of intelligence. Date records source review, not evaluation.","ok":true,"startedAt":"2026-09-06T00:00:00Z"},{"adapter":"launch-page-review","finishedAt":"2026-09-06T00:00:00Z","id":"launch-openai-astra-system-card","note":"GPT-6 Astra System Card. Date records source review, not evaluation.","ok":true,"startedAt":"2026-09-06T00:00:00Z"},{"adapter":"launch-page-review","finishedAt":"2026-09-22T00:00:00Z","id":"launch-openai-gpt6-sol-luna-launch","note":"Introducing GPT-6 Sol and Luna. Date records source review, not evaluation.","ok":true,"startedAt":"2026-09-22T00:00:00Z"},{"adapter":"launch-page-review","finishedAt":"2026-09-29T00:00:00Z","id":"launch-openai-gpt61-sol-launch","note":"Introducing GPT-6.1 Sol. Date records source review, not evaluation.","ok":true,"startedAt":"2026-09-29T00:00:00Z"},{"adapter":"launch-page-review","finishedAt":"2026-09-06T00:00:00Z","id":"launch-openai-sol-launch","note":"GPT-5.6: Frontier intelligence that scales with your ambition. Date records source review, not evaluation.","ok":true,"startedAt":"2026-09-06T00:00:00Z"},{"adapter":"launch-page-review","finishedAt":"2026-09-30T00:00:00Z","id":"launch-qwen","note":"Qwen/Qwen3.8-2.4T-A95B. Date records source review, not evaluation.","ok":true,"startedAt":"2026-09-30T00:00:00Z"},{"adapter":"launch-page-review","finishedAt":"2026-09-06T00:00:00Z","id":"launch-xai","note":"Grok 4.6 launch evaluations. Date records source review, not evaluation.","ok":true,"startedAt":"2026-09-06T00:00:00Z"},{"adapter":"launch-page-review","finishedAt":"2026-09-21T00:00:00Z","id":"launch-xai-grok47","note":"Grok 4.7 launch comparison. Date records source review, not evaluation.","ok":true,"startedAt":"2026-09-21T00:00:00Z"},{"adapter":"launch-page-review","finishedAt":"2026-09-30T00:00:00Z","id":"launch-zai","note":"zai-org/GLM-5.3. Date records source review, not evaluation.","ok":true,"startedAt":"2026-09-30T00:00:00Z"}],"labs":[{"id":"openai","name":"OpenAI","url":"https://openai.com"},{"id":"anthropic","name":"Anthropic","url":"https://www.anthropic.com"},{"id":"xai","name":"xAI","url":"https://x.ai"},{"id":"google","name":"Google","url":"https://deepmind.google"},{"id":"meta","name":"Meta","url":"https://ai.meta.com"},{"id":"mistral","name":"Mistral","url":"https://mistral.ai"},{"id":"moonshot","name":"Moonshot","url":"https://www.kimi.ai"},{"id":"qwen","name":"Qwen","url":"https://qwen.ai"},{"id":"deepseek","name":"DeepSeek","url":"https://www.deepseek.com"},{"id":"nvidia","name":"NVIDIA","url":"https://www.nvidia.com"},{"id":"zai","name":"Z.ai","url":"https://z.ai/"},{"id":"minimax","name":"MiniMax","url":"https://www.minimax.io/"},{"id":"cohere","name":"Cohere","url":"https://cohere.com/"},{"id":"amazon","name":"Amazon","url":"https://aws.amazon.com/nova/"},{"id":"microsoft","name":"Microsoft","url":"https://microsoft.ai/"},{"id":"xiaomi","name":"Xiaomi","url":"https://mimo.mi.com"},{"id":"stepfun","name":"StepFun","url":"https://www.stepfun.com"},{"id":"lgai","name":"LG AI Research","url":"https://www.lgresearch.ai"},{"id":"upstage","name":"Upstage","url":"https://www.upstage.ai"},{"id":"bytedance","name":"ByteDance","url":"https://seed.bytedance.com"},{"id":"tencent","name":"Tencent","url":"https://hunyuan.tencent.com"},{"id":"baidu","name":"Baidu","url":"https://ernie.baidu.com"},{"id":"openbmb","name":"OpenBMB","url":"https://huggingface.co/openbmb"},{"id":"agnes-ai","name":"Agnes AI","url":"https://agnes-ai.com/"}],"modelAliases":[{"alias":"gpt-5.6","modelId":"gpt-5.6-sol"},{"alias":"GPT-5.6 Sol","modelId":"gpt-5.6-sol"},{"alias":"GPT-5.6 Sol (max)","modelId":"gpt-5.6-sol"},{"alias":"gpt-5-6-sol","modelId":"gpt-5.6-sol"},{"alias":"gpt-5.6-sol-xhigh","modelId":"gpt-5.6-sol"},{"alias":"openai-gpt-5-6-sol-max","modelId":"gpt-5.6-sol"},{"alias":"opus-5","modelId":"claude-opus-5"},{"alias":"Claude Opus 5","modelId":"claude-opus-5"},{"alias":"claude-opus-5-high","modelId":"claude-opus-5"},{"alias":"Claude Opus 5 (Adaptive Reasoning, Max Effort)","modelId":"claude-opus-5"},{"alias":"Claude Opus 5 (max)","modelId":"claude-opus-5"},{"alias":"anthropic-claude-opus-5-max","modelId":"claude-opus-5"},{"alias":"grok-4.6","modelId":"grok-4.6"},{"alias":"grok-4.6-high","modelId":"grok-4.6"},{"alias":"Grok 4.6 (high)","modelId":"grok-4.6"},{"alias":"grok-4-6","modelId":"grok-4.6"},{"alias":"xai-grok-4-6-high","modelId":"grok-4.6"},{"alias":"gemini-3.1-pro","modelId":"gemini-3.1-pro"},{"alias":"gemini-3.1-pro (thinking)","modelId":"gemini-3.1-pro"},{"alias":"gemini-3.1-pro-preview","modelId":"gemini-3.1-pro"},{"alias":"gemini-3.1-pro-preview (thinking high)","modelId":"gemini-3.1-pro"},{"alias":"Gemini 3.1 Pro Preview","modelId":"gemini-3.1-pro"},{"alias":"Gemini 3.1 Pro (Preview)","modelId":"gemini-3.1-pro"},{"alias":"gemini-3-1-pro-preview","modelId":"gemini-3.1-pro"},{"alias":"Llama 4 Maverick","modelId":"llama-4-maverick"},{"alias":"Llama 4 Maverick Instruct","modelId":"llama-4-maverick"},{"alias":"llama-4-maverick-instruct","modelId":"llama-4-maverick"},{"alias":"llama-4-maverick-17b-128e-instruct","modelId":"llama-4-maverick"},{"alias":"Llama4-Maverick","modelId":"llama-4-maverick"},{"alias":"mistral-large-2512","modelId":"mistral-large-3"},{"alias":"kimi-k3","modelId":"kimi-k3"},{"alias":"kimi-k3-max","modelId":"kimi-k3"},{"alias":"Kimi K3 (max)","modelId":"kimi-k3"},{"alias":"moonshot-kimi-k3-max","modelId":"kimi-k3"},{"alias":"qwen3.8-max","modelId":"qwen-3.8-max"},{"alias":"Qwen3.8 Max","modelId":"qwen-3.8-max"},{"alias":"qwen3-8-max","modelId":"qwen-3.8-max"},{"alias":"DeepSeek-V4-Pro-0813","modelId":"deepseek-v4-pro"},{"alias":"DeepSeek V4 Pro 0813","modelId":"deepseek-v4-pro"},{"alias":"DeepSeek V4 Pro 0813 (Reasoning, Max Effort)","modelId":"deepseek-v4-pro"},{"alias":"DeepSeek V4 Pro 0813 (max)","modelId":"deepseek-v4-pro"},{"alias":"deepseek-v4-pro-0813-max","modelId":"deepseek-v4-pro"},{"alias":"nemotron-3-ultra","modelId":"nemotron-3-ultra"},{"alias":"Nemotron 3 Ultra","modelId":"nemotron-3-ultra"},{"alias":"Nemotron 3 Ultra 550B A55B (Reasoning)","modelId":"nemotron-3-ultra"},{"alias":"nvidia-nemotron-3-ultra-550b-a55b-nvfp4","modelId":"nemotron-3-ultra"},{"alias":"nvidia-nemotron-3-ultra-550b-a55b","modelId":"nemotron-3-ultra"},{"alias":"Claude Opus 5.5","modelId":"claude-opus-5-5"},{"alias":"anthropic.claude-opus-5-5","modelId":"claude-opus-5-5"},{"alias":"deepseek-ai/DeepSeek-R1-0528","modelId":"deepseek-r1-0528"},{"alias":"deepseek-ai/DeepSeek-R1-0528-Qwen3-8B","modelId":"deepseek-r1-0528-qwen3-8b"},{"alias":"deepseek-ai/DeepSeek-R1-Distill-Llama-70B","modelId":"deepseek-r1-distill-llama-70b"},{"alias":"deepseek-ai/DeepSeek-R1-Distill-Llama-8B","modelId":"deepseek-r1-distill-llama-8b"},{"alias":"deepseek-ai/DeepSeek-R1-Distill-Qwen-1.5B","modelId":"deepseek-r1-distill-qwen-1.5b"},{"alias":"deepseek-ai/DeepSeek-R1-Distill-Qwen-14B","modelId":"deepseek-r1-distill-qwen-14b"},{"alias":"deepseek-ai/DeepSeek-R1-Distill-Qwen-32B","modelId":"deepseek-r1-distill-qwen-32b"},{"alias":"deepseek-ai/DeepSeek-R1-Distill-Qwen-7B","modelId":"deepseek-r1-distill-qwen-7b"},{"alias":"deepseek-ai/DeepSeek-V3.2","modelId":"deepseek-v3.2"},{"alias":"deepseek-ai/DeepSeek-V3.2-Speciale","modelId":"deepseek-v3.2-speciale"},{"alias":"deepseek-ai/DeepSeek-V4-Flash-0731","modelId":"deepseek-v4-flash"},{"alias":"deepseek-v4-flash-0731","modelId":"deepseek-v4-flash"},{"alias":"deepseek-ai/DeepSeek-V4-Flash-Vision-Exp","modelId":"deepseek-v4-flash-vision-exp"},{"alias":"deepseek-ai/DeepSeek-V4-Pro-0813","modelId":"deepseek-v4-pro"},{"alias":"meta-llama/Llama-3.1-405B-Instruct","modelId":"llama-3.1-405b-instruct"},{"alias":"meta-llama/Llama-3.1-70B-Instruct","modelId":"llama-3.1-70b-instruct"},{"alias":"meta-llama/Llama-3.1-8B-Instruct","modelId":"llama-3.1-8b-instruct"},{"alias":"meta-llama/Llama-3.2-11B-Vision-Instruct","modelId":"llama-3.2-11b-vision-instruct"},{"alias":"meta-llama/Llama-3.2-1B-Instruct","modelId":"llama-3.2-1b-instruct"},{"alias":"meta-llama/Llama-3.2-3B-Instruct","modelId":"llama-3.2-3b-instruct"},{"alias":"meta-llama/Llama-3.2-90B-Vision-Instruct","modelId":"llama-3.2-90b-vision-instruct"},{"alias":"meta-llama/Llama-3.3-70B-Instruct","modelId":"llama-3.3-70b-instruct"},{"alias":"meta-llama/Llama-4-Maverick-17B-128E-Instruct","modelId":"llama-4-maverick"},{"alias":"meta-llama/Llama-4-Scout-17B-16E-Instruct","modelId":"llama-4-scout"},{"alias":"llama-4-scout-17b-16e-instruct","modelId":"llama-4-scout"},{"alias":"meta-models/Muse-Glimmer-30B","modelId":"muse-glimmer-30b"},{"alias":"labs-leanstral-1-5","modelId":"leanstral-1.5"},{"alias":"mistral-medium-3-5","modelId":"mistral-medium-3.5"},{"alias":"mistral-small-2603","modelId":"mistral-small-4"},{"alias":"moonshotai/Kimi-Dev-72B","modelId":"kimi-dev-72b"},{"alias":"moonshotai/Kimi-K2.6","modelId":"kimi-k2.6"},{"alias":"moonshotai/Kimi-K2.7-Code","modelId":"kimi-k2.7-code"},{"alias":"kimi-k2.7-code-highspeed","modelId":"kimi-k2.7-code"},{"alias":"moonshotai/Kimi-K3","modelId":"kimi-k3"},{"alias":"moonshotai/Kimi-Linear-48B-A3B-Instruct","modelId":"kimi-linear-48b-a3b-instruct"},{"alias":"moonshotai/Kimi-VL-A3B-Instruct","modelId":"kimi-vl-a3b-instruct"},{"alias":"moonshotai/Kimi-VL-A3B-Thinking-2506","modelId":"kimi-vl-a3b-thinking-2506"},{"alias":"nvidia/Nemotron-3-Nano-Omni-30B-A3B-Reasoning-BF16","modelId":"nemotron-3-nano-omni-30b-a3b-reasoning-bf16"},{"alias":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","modelId":"nemotron-3-ultra"},{"alias":"nvidia-nemotron-3-ultra-550b-a55b-bf16","modelId":"nemotron-3-ultra"},{"alias":"nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-BF16","modelId":"nvidia-nemotron-3-nano-30b-a3b-bf16"},{"alias":"nvidia/NVIDIA-Nemotron-3-Nano-4B-BF16","modelId":"nvidia-nemotron-3-nano-4b-bf16"},{"alias":"nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16","modelId":"nvidia-nemotron-3-super-120b-a12b-bf16"},{"alias":"nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16","modelId":"nvidia-nemotron-3.5-lightning-30b-a3b-bf16"},{"alias":"nvidia/NVIDIA-Nemotron-Nano-12B-v2","modelId":"nvidia-nemotron-nano-12b-v2"},{"alias":"nvidia/NVIDIA-Nemotron-Nano-12B-v2-VL-BF16","modelId":"nvidia-nemotron-nano-12b-v2-vl-bf16"},{"alias":"nvidia/NVIDIA-Nemotron-Nano-9B-v2","modelId":"nvidia-nemotron-nano-9b-v2"},{"alias":"nvidia/NVIDIA-Nemotron-Nano-9B-v2-Japanese","modelId":"nvidia-nemotron-nano-9b-v2-japanese"},{"alias":"GPT-6 Luna","modelId":"gpt-6-luna"},{"alias":"GPT-6.1 Sol","modelId":"gpt-6.1-sol"},{"alias":"GPT-6 Sol","modelId":"gpt-6-sol"},{"alias":"Qwen/Qwen3-0.6B","modelId":"qwen3-0.6b"},{"alias":"Qwen/Qwen3-1.7B","modelId":"qwen3-1.7b"},{"alias":"Qwen/Qwen3-14B","modelId":"qwen3-14b"},{"alias":"Qwen/Qwen3-235B-A22B","modelId":"qwen3-235b-a22b"},{"alias":"Qwen/Qwen3-235B-A22B-Instruct-2507","modelId":"qwen3-235b-a22b-instruct-2507"},{"alias":"Qwen/Qwen3-235B-A22B-Thinking-2507","modelId":"qwen3-235b-a22b-thinking-2507"},{"alias":"Qwen/Qwen3-30B-A3B","modelId":"qwen3-30b-a3b"},{"alias":"Qwen/Qwen3-30B-A3B-Instruct-2507","modelId":"qwen3-30b-a3b-instruct-2507"},{"alias":"Qwen/Qwen3-30B-A3B-Thinking-2507","modelId":"qwen3-30b-a3b-thinking-2507"},{"alias":"Qwen/Qwen3-32B","modelId":"qwen3-32b"},{"alias":"Qwen/Qwen3-4B","modelId":"qwen3-4b"},{"alias":"Qwen/Qwen3-4B-Instruct-2507","modelId":"qwen3-4b-instruct-2507"},{"alias":"Qwen/Qwen3-4B-Thinking-2507","modelId":"qwen3-4b-thinking-2507"},{"alias":"Qwen/Qwen3-8B","modelId":"qwen3-8b"},{"alias":"Qwen/Qwen3-Coder-30B-A3B-Instruct","modelId":"qwen3-coder-30b-a3b-instruct"},{"alias":"Qwen/Qwen3-Coder-480B-A35B-Instruct","modelId":"qwen3-coder-480b-a35b-instruct"},{"alias":"Qwen/Qwen3-Coder-Next","modelId":"qwen3-coder-next"},{"alias":"Qwen/Qwen3-Next-80B-A3B-Instruct","modelId":"qwen3-next-80b-a3b-instruct"},{"alias":"Qwen/Qwen3-Next-80B-A3B-Thinking","modelId":"qwen3-next-80b-a3b-thinking"},{"alias":"Qwen/Qwen3-Omni-30B-A3B-Instruct","modelId":"qwen3-omni-30b-a3b-instruct"},{"alias":"Qwen/Qwen3-Omni-30B-A3B-Thinking","modelId":"qwen3-omni-30b-a3b-thinking"},{"alias":"Qwen/Qwen3-VL-235B-A22B-Instruct","modelId":"qwen3-vl-235b-a22b-instruct"},{"alias":"Qwen/Qwen3-VL-235B-A22B-Thinking","modelId":"qwen3-vl-235b-a22b-thinking"},{"alias":"Qwen/Qwen3-VL-2B-Instruct","modelId":"qwen3-vl-2b-instruct"},{"alias":"Qwen/Qwen3-VL-2B-Thinking","modelId":"qwen3-vl-2b-thinking"},{"alias":"Qwen/Qwen3-VL-30B-A3B-Instruct","modelId":"qwen3-vl-30b-a3b-instruct"},{"alias":"Qwen/Qwen3-VL-30B-A3B-Thinking","modelId":"qwen3-vl-30b-a3b-thinking"},{"alias":"Qwen/Qwen3-VL-32B-Instruct","modelId":"qwen3-vl-32b-instruct"},{"alias":"Qwen/Qwen3-VL-32B-Thinking","modelId":"qwen3-vl-32b-thinking"},{"alias":"Qwen/Qwen3-VL-4B-Instruct","modelId":"qwen3-vl-4b-instruct"},{"alias":"Qwen/Qwen3-VL-4B-Thinking","modelId":"qwen3-vl-4b-thinking"},{"alias":"Qwen/Qwen3-VL-8B-Instruct","modelId":"qwen3-vl-8b-instruct"},{"alias":"Qwen/Qwen3-VL-8B-Thinking","modelId":"qwen3-vl-8b-thinking"},{"alias":"Qwen/Qwen3.5-0.8B","modelId":"qwen3.5-0.8b"},{"alias":"Qwen/Qwen3.5-122B-A10B","modelId":"qwen3.5-122b-a10b"},{"alias":"Qwen/Qwen3.5-27B","modelId":"qwen3.5-27b"},{"alias":"Qwen/Qwen3.5-2B","modelId":"qwen3.5-2b"},{"alias":"Qwen/Qwen3.5-35B-A3B","modelId":"qwen3.5-35b-a3b"},{"alias":"Qwen/Qwen3.5-397B-A17B","modelId":"qwen3.5-397b-a17b"},{"alias":"Qwen/Qwen3.5-4B","modelId":"qwen3.5-4b"},{"alias":"Qwen/Qwen3.5-9B","modelId":"qwen3.5-9b"},{"alias":"Qwen/Qwen3.6-27B","modelId":"qwen3.6-27b"},{"alias":"Qwen/Qwen3.6-35B-A3B","modelId":"qwen3.6-35b-a3b"},{"alias":"Qwen/Qwen3.8-2.4T-A95B","modelId":"qwen3.8-2.4t-a95b"},{"alias":"Qwen/Qwen3.8-27B","modelId":"qwen3.8-27b"},{"alias":"Qwen/Qwen3.8-Flash-Next","modelId":"qwen3.8-flash-next"},{"alias":"Grok-4.7","modelId":"grok-4.7"},{"alias":"grok-4-7","modelId":"grok-4.7"},{"alias":"zai-org/GLM-5.3-Flash","modelId":"glm-5.3-flash"},{"alias":"zai-org/GLM-5.3","modelId":"glm-5.3"},{"alias":"zai-org/GLM-5.2","modelId":"glm-5.2"},{"alias":"zai-org/GLM-5.1","modelId":"glm-5.1"},{"alias":"zai-org/GLM-5","modelId":"glm-5"},{"alias":"zai-org/GLM-4.7","modelId":"glm-4.7"},{"alias":"zai-org/GLM-4.6","modelId":"glm-4.6"},{"alias":"zai-org/GLM-4.5","modelId":"glm-4.5"},{"alias":"zai-org/GLM-4.5-Air","modelId":"glm-4.5-air"},{"alias":"zai-org/GLM-4.7-Flash","modelId":"glm-4.7-flash"},{"alias":"zai-org/GLM-4.6V","modelId":"glm-4.6v"},{"alias":"zai-org/GLM-4.5V","modelId":"glm-4.5v"},{"alias":"zai-org/GLM-4.6V-Flash","modelId":"glm-4.6v-flash"},{"alias":"zai-org/GLM-4.1V-9B-Thinking","modelId":"glm-4.1v-9b-thinking"},{"alias":"zai-org/GLM-Z1-Rumination-32B-0414","modelId":"glm-z1-rumination-32b-0414"},{"alias":"zai-org/GLM-Z1-9B-0414","modelId":"glm-z1-9b-0414"},{"alias":"zai-org/GLM-Z1-32B-0414","modelId":"glm-z1-32b-0414"},{"alias":"zai-org/GLM-4-9B-0414","modelId":"glm-4-9b-0414"},{"alias":"zai-org/GLM-4-32B-0414","modelId":"glm-4-32b-0414"},{"alias":"zai-org/glm-4-9b-chat","modelId":"glm-4-9b-chat"},{"alias":"zai-org/glm-4-9b-chat-1m","modelId":"glm-4-9b-chat-1m"},{"alias":"zai-org/glm-4v-9b","modelId":"glm-4v-9b"},{"alias":"zai-org/glm-edge-1.5b-chat","modelId":"glm-edge-1.5b-chat"},{"alias":"zai-org/glm-edge-4b-chat","modelId":"glm-edge-4b-chat"},{"alias":"zai-org/glm-edge-v-2b","modelId":"glm-edge-v-2b"},{"alias":"zai-org/glm-edge-v-5b","modelId":"glm-edge-v-5b"},{"alias":"zai-org/chatglm-6b","modelId":"chatglm-6b"},{"alias":"zai-org/chatglm2-6b","modelId":"chatglm2-6b"},{"alias":"zai-org/chatglm2-6b-32k","modelId":"chatglm2-6b-32k"},{"alias":"zai-org/chatglm3-6b","modelId":"chatglm3-6b"},{"alias":"zai-org/chatglm3-6b-32k","modelId":"chatglm3-6b-32k"},{"alias":"zai-org/chatglm3-6b-128k","modelId":"chatglm3-6b-128k"},{"alias":"zai-org/codegeex4-all-9b","modelId":"codegeex4-all-9b"},{"alias":"zai-org/AutoGLM-Phone-9B","modelId":"autoglm-phone-9b"},{"alias":"zai-org/AutoGLM-Phone-9B-Multilingual","modelId":"autoglm-phone-9b-multilingual"},{"alias":"zai-org/cogagent-9b-20241220","modelId":"cogagent-9b-20241220"},{"alias":"zai-org/codegeex2-6b","modelId":"codegeex2-6b"},{"alias":"zai-org/visualglm-6b","modelId":"visualglm-6b"},{"alias":"zai-org/cogvlm-chat-hf","modelId":"cogvlm-chat-hf"},{"alias":"zai-org/cogagent-chat-hf","modelId":"cogagent-chat-hf"},{"alias":"zai-org/cogagent-vqa-hf","modelId":"cogagent-vqa-hf"},{"alias":"zai-org/cogvlm2-llama3-chat-19B","modelId":"cogvlm2-llama3-chat-19b"},{"alias":"zai-org/cogvlm2-llama3-chinese-chat-19B","modelId":"cogvlm2-llama3-chinese-chat-19b"},{"alias":"zai-org/cogvlm2-video-llama3-chat","modelId":"cogvlm2-video-llama3-chat"},{"alias":"MiniMaxAI/MiniMax-M3","modelId":"minimax-m3"},{"alias":"MiniMax-M2.7-highspeed","modelId":"minimax-m2.7"},{"alias":"MiniMaxAI/MiniMax-M2.7","modelId":"minimax-m2.7"},{"alias":"MiniMax-M2.5-highspeed","modelId":"minimax-m2.5"},{"alias":"MiniMaxAI/MiniMax-M2.5","modelId":"minimax-m2.5"},{"alias":"MiniMax-M2.1-highspeed","modelId":"minimax-m2.1"},{"alias":"MiniMaxAI/MiniMax-M2.1","modelId":"minimax-m2.1"},{"alias":"MiniMaxAI/MiniMax-M2","modelId":"minimax-m2"},{"alias":"MiniMaxAI/MiniMax-Text-01","modelId":"minimax-text-01"},{"alias":"MiniMaxAI/MiniMax-VL-01","modelId":"minimax-vl-01"},{"alias":"MiniMaxAI/MiniMax-M1-40k","modelId":"minimax-m1-40k"},{"alias":"MiniMaxAI/MiniMax-M1-80k","modelId":"minimax-m1-80k"},{"alias":"MiniMaxAI/SynLogic-32B","modelId":"synlogic-32b"},{"alias":"MiniMaxAI/SynLogic-Mix-3-32B","modelId":"synlogic-mix-3-32b"},{"alias":"MiniMaxAI/SynLogic-7B","modelId":"synlogic-7b"},{"alias":"M2-her","modelId":"minimax-m2-her"},{"alias":"CohereLabs/command-a-plus-05-2026-bf16","modelId":"command-a-plus-05-2026"},{"alias":"CohereLabs/c4ai-command-a-03-2025","modelId":"command-a-03-2025"},{"alias":"CohereLabs/c4ai-command-r7b-12-2024","modelId":"command-r7b-12-2024"},{"alias":"CohereLabs/command-a-translate-08-2025","modelId":"command-a-translate-08-2025"},{"alias":"CohereLabs/command-a-reasoning-08-2025","modelId":"command-a-reasoning-08-2025"},{"alias":"CohereLabs/command-a-vision-07-2025","modelId":"command-a-vision-07-2025"},{"alias":"CohereLabs/c4ai-command-r-08-2024","modelId":"command-r-08-2024"},{"alias":"CohereLabs/c4ai-command-r-plus-08-2024","modelId":"command-r-plus-08-2024"},{"alias":"command-r","modelId":"command-r-03-2024"},{"alias":"CohereLabs/c4ai-command-r-v01","modelId":"command-r-03-2024"},{"alias":"command-r-plus","modelId":"command-r-plus-04-2024"},{"alias":"CohereLabs/c4ai-command-r-plus","modelId":"command-r-plus-04-2024"},{"alias":"CohereLabs/tiny-aya-global","modelId":"tiny-aya-global"},{"alias":"CohereLabs/tiny-aya-earth","modelId":"tiny-aya-earth"},{"alias":"CohereLabs/tiny-aya-fire","modelId":"tiny-aya-fire"},{"alias":"CohereLabs/tiny-aya-water","modelId":"tiny-aya-water"},{"alias":"CohereLabs/aya-expanse-32b","modelId":"c4ai-aya-expanse-32b"},{"alias":"CohereLabs/aya-vision-32b","modelId":"c4ai-aya-vision-32b"},{"alias":"CohereLabs/c4ai-command-r7b-arabic-02-2025","modelId":"c4ai-command-r7b-arabic-02-2025"},{"alias":"CohereLabs/aya-expanse-8b","modelId":"c4ai-aya-expanse-8b"},{"alias":"CohereLabs/aya-vision-8b","modelId":"c4ai-aya-vision-8b"},{"alias":"CohereLabs/aya-23-8B","modelId":"aya-23-8b"},{"alias":"CohereLabs/aya-23-35B","modelId":"aya-23-35b"},{"alias":"CohereLabs/aya-101","modelId":"aya-101"},{"alias":"CohereLabs/North-Mini-Code-1.0","modelId":"north-mini-code-1-0"},{"alias":"CohereLabs/North-Micro-Vision-Instruct","modelId":"north-micro-vision-instruct"},{"alias":"microsoft/phi-4","modelId":"phi-4"},{"alias":"microsoft/Phi-4-mini-instruct","modelId":"phi-4-mini-instruct"},{"alias":"microsoft/Phi-4-reasoning-vision-15B","modelId":"phi-4-reasoning-vision-15b"},{"alias":"microsoft/Phi-3-mini-4k-instruct","modelId":"phi-3-mini-4k-instruct"},{"alias":"microsoft/Phi-3-mini-128k-instruct","modelId":"phi-3-mini-128k-instruct"},{"alias":"microsoft/Phi-3.5-mini-instruct","modelId":"phi-3.5-mini-instruct"},{"alias":"microsoft/Phi-4-multimodal-instruct","modelId":"phi-4-multimodal-instruct"},{"alias":"microsoft/Phi-4-reasoning","modelId":"phi-4-reasoning"},{"alias":"microsoft/Phi-mini-MoE-instruct","modelId":"phi-mini-moe-instruct"},{"alias":"microsoft/Phi-tiny-MoE-instruct","modelId":"phi-tiny-moe-instruct"},{"alias":"microsoft/Phi-3-medium-4k-instruct","modelId":"phi-3-medium-4k-instruct"},{"alias":"microsoft/Phi-3-medium-128k-instruct","modelId":"phi-3-medium-128k-instruct"},{"alias":"microsoft/Phi-3-small-8k-instruct","modelId":"phi-3-small-8k-instruct"},{"alias":"microsoft/Phi-3-small-128k-instruct","modelId":"phi-3-small-128k-instruct"},{"alias":"microsoft/Phi-3-vision-128k-instruct","modelId":"phi-3-vision-128k-instruct"},{"alias":"microsoft/Phi-3.5-vision-instruct","modelId":"phi-3.5-vision-instruct"},{"alias":"microsoft/Phi-3.5-MoE-instruct","modelId":"phi-3.5-moe-instruct"},{"alias":"microsoft/Phi-4-reasoning-plus","modelId":"phi-4-reasoning-plus"},{"alias":"microsoft/Phi-4-mini-reasoning","modelId":"phi-4-mini-reasoning"},{"alias":"microsoft/Phi-4-mini-flash-reasoning","modelId":"phi-4-mini-flash-reasoning"},{"alias":"microsoft/MAI-DS-R1","modelId":"mai-ds-r1"},{"alias":"MAI-Code-1.1-Flash","modelId":"mai-code-1-1-flash"},{"alias":"amazon.nova-micro-v1:0","modelId":"amazon-nova-micro"},{"alias":"amazon.nova-lite-v1:0","modelId":"amazon-nova-lite"},{"alias":"amazon.nova-pro-v1:0","modelId":"amazon-nova-pro"},{"alias":"amazon.nova-premier-v1:0","modelId":"amazon-nova-premier"},{"alias":"amazon.nova-2-lite-v1:0","modelId":"amazon-nova-2-lite"},{"alias":"global.amazon.nova-2-lite-v1:0","modelId":"amazon-nova-2-lite"},{"alias":"us.amazon.nova-2-lite-v1:0","modelId":"amazon-nova-2-lite"},{"alias":"DeepSeek V4 Pro 0424","modelId":"deepseek-v4-pro-0424"},{"alias":"deepseek-v4-pro-0424","modelId":"deepseek-v4-pro-0424"},{"alias":"ByteDance-Seed/Seed-Coder-8B-Instruct","modelId":"seed-coder-8b-instruct"},{"alias":"Seed-Coder-8B-Instruct","modelId":"seed-coder-8b-instruct"},{"alias":"ByteDance-Seed/Seed-Coder-8B-Reasoning","modelId":"seed-coder-8b-reasoning"},{"alias":"Seed-Coder-8B-Reasoning","modelId":"seed-coder-8b-reasoning"},{"alias":"ByteDance-Seed/Seed-OSS-36B-Instruct","modelId":"seed-oss-36b-instruct"},{"alias":"Seed-OSS-36B-Instruct","modelId":"seed-oss-36b-instruct"},{"alias":"LGAI-EXAONE/EXAONE-3.0-7.8B-Instruct","modelId":"exaone-3.0-7.8b-instruct"},{"alias":"EXAONE-3.0-7.8B-Instruct","modelId":"exaone-3.0-7.8b-instruct"},{"alias":"LGAI-EXAONE/EXAONE-3.5-2.4B-Instruct","modelId":"exaone-3.5-2.4b-instruct"},{"alias":"EXAONE-3.5-2.4B-Instruct","modelId":"exaone-3.5-2.4b-instruct"},{"alias":"LGAI-EXAONE/EXAONE-3.5-32B-Instruct","modelId":"exaone-3.5-32b-instruct"},{"alias":"EXAONE-3.5-32B-Instruct","modelId":"exaone-3.5-32b-instruct"},{"alias":"LGAI-EXAONE/EXAONE-3.5-7.8B-Instruct","modelId":"exaone-3.5-7.8b-instruct"},{"alias":"EXAONE-3.5-7.8B-Instruct","modelId":"exaone-3.5-7.8b-instruct"},{"alias":"LGAI-EXAONE/EXAONE-4.0-1.2B","modelId":"exaone-4.0-1.2b"},{"alias":"EXAONE-4.0-1.2B","modelId":"exaone-4.0-1.2b"},{"alias":"LGAI-EXAONE/EXAONE-4.0-32B","modelId":"exaone-4.0-32b"},{"alias":"EXAONE-4.0-32B","modelId":"exaone-4.0-32b"},{"alias":"LGAI-EXAONE/EXAONE-4.0.1-32B","modelId":"exaone-4.0.1-32b"},{"alias":"EXAONE-4.0.1-32B","modelId":"exaone-4.0.1-32b"},{"alias":"LGAI-EXAONE/EXAONE-4.5-33B","modelId":"exaone-4.5-33b"},{"alias":"EXAONE-4.5-33B","modelId":"exaone-4.5-33b"},{"alias":"LGAI-EXAONE/EXAONE-Deep-2.4B","modelId":"exaone-deep-2.4b"},{"alias":"EXAONE-Deep-2.4B","modelId":"exaone-deep-2.4b"},{"alias":"LGAI-EXAONE/EXAONE-Deep-32B","modelId":"exaone-deep-32b"},{"alias":"EXAONE-Deep-32B","modelId":"exaone-deep-32b"},{"alias":"LGAI-EXAONE/EXAONE-Deep-7.8B","modelId":"exaone-deep-7.8b"},{"alias":"EXAONE-Deep-7.8B","modelId":"exaone-deep-7.8b"},{"alias":"LGAI-EXAONE/K-EXAONE-2.0-750B-A37B","modelId":"k-exaone-2.0-750b-a37b"},{"alias":"K-EXAONE-2.0-750B-A37B","modelId":"k-exaone-2.0-750b-a37b"},{"alias":"LGAI-EXAONE/K-EXAONE-236B-A23B","modelId":"k-exaone-236b-a23b"},{"alias":"K-EXAONE-236B-A23B","modelId":"k-exaone-236b-a23b"},{"alias":"stepfun-ai/Step-3.5-Flash","modelId":"step-3.5-flash"},{"alias":"Step-3.5-Flash","modelId":"step-3.5-flash"},{"alias":"stepfun-ai/Step-3.7-Flash","modelId":"step-3.7-flash"},{"alias":"Step-3.7-Flash","modelId":"step-3.7-flash"},{"alias":"stepfun-ai/step3","modelId":"step3"},{"alias":"step3","modelId":"step3"},{"alias":"stepfun-ai/Step3-VL-10B","modelId":"step3-vl-10b"},{"alias":"Step3-VL-10B","modelId":"step3-vl-10b"},{"alias":"upstage/SOLAR-10.7B-Instruct-v1.0","modelId":"solar-10.7b-instruct-v1.0"},{"alias":"SOLAR-10.7B-Instruct-v1.0","modelId":"solar-10.7b-instruct-v1.0"},{"alias":"solar-mini","modelId":"solar-mini-250422"},{"alias":"upstage/Solar-Open-100B","modelId":"solar-open-100b"},{"alias":"Solar-Open-100B","modelId":"solar-open-100b"},{"alias":"upstage/Solar-Open2-250B","modelId":"solar-open2-250b"},{"alias":"Solar-Open2-250B","modelId":"solar-open2-250b"},{"alias":"upstage/solar-pro-preview-instruct","modelId":"solar-pro-preview-instruct"},{"alias":"solar-pro-preview-instruct","modelId":"solar-pro-preview-instruct"},{"alias":"solar-pro2","modelId":"solar-pro2-251215"},{"alias":"solar-pro3","modelId":"solar-pro3-260323"},{"alias":"solar-pro4","modelId":"solar-pro4-260806"},{"alias":"XiaomiMiMo/MiMo-7B-RL","modelId":"mimo-7b-rl"},{"alias":"MiMo-7B-RL","modelId":"mimo-7b-rl"},{"alias":"XiaomiMiMo/MiMo-7B-RL-0530","modelId":"mimo-7b-rl-0530"},{"alias":"MiMo-7B-RL-0530","modelId":"mimo-7b-rl-0530"},{"alias":"XiaomiMiMo/MiMo-7B-RL-Zero","modelId":"mimo-7b-rl-zero"},{"alias":"MiMo-7B-RL-Zero","modelId":"mimo-7b-rl-zero"},{"alias":"XiaomiMiMo/MiMo-7B-SFT","modelId":"mimo-7b-sft"},{"alias":"MiMo-7B-SFT","modelId":"mimo-7b-sft"},{"alias":"XiaomiMiMo/MiMo-V2-Flash","modelId":"mimo-v2-flash"},{"alias":"MiMo-V2-Flash","modelId":"mimo-v2-flash"},{"alias":"XiaomiMiMo/MiMo-V2.5","modelId":"mimo-v2.5"},{"alias":"MiMo-V2.5","modelId":"mimo-v2.5"},{"alias":"XiaomiMiMo/MiMo-V2.5-Pro","modelId":"mimo-v2.5-pro"},{"alias":"MiMo-V2.5-Pro","modelId":"mimo-v2.5-pro"},{"alias":"XiaomiMiMo/MiMo-VL-7B-RL","modelId":"mimo-vl-7b-rl"},{"alias":"MiMo-VL-7B-RL","modelId":"mimo-vl-7b-rl"},{"alias":"XiaomiMiMo/MiMo-VL-7B-RL-2508","modelId":"mimo-vl-7b-rl-2508"},{"alias":"MiMo-VL-7B-RL-2508","modelId":"mimo-vl-7b-rl-2508"},{"alias":"XiaomiMiMo/MiMo-VL-7B-SFT","modelId":"mimo-vl-7b-sft"},{"alias":"MiMo-VL-7B-SFT","modelId":"mimo-vl-7b-sft"},{"alias":"XiaomiMiMo/MiMo-VL-7B-SFT-2508","modelId":"mimo-vl-7b-sft-2508"},{"alias":"MiMo-VL-7B-SFT-2508","modelId":"mimo-vl-7b-sft-2508"},{"alias":"baidu/ERNIE-4.5-0.3B-PT","modelId":"ernie-4.5-0.3b-pt"},{"alias":"baidu/ERNIE-4.5-21B-A3B-PT","modelId":"ernie-4.5-21b-a3b-pt"},{"alias":"baidu/ERNIE-4.5-21B-A3B-Thinking","modelId":"ernie-4.5-21b-a3b-thinking"},{"alias":"baidu/ERNIE-4.5-300B-A47B-PT","modelId":"ernie-4.5-300b-a47b-pt"},{"alias":"baidu/ERNIE-4.5-VL-28B-A3B-PT","modelId":"ernie-4.5-vl-28b-a3b-pt"},{"alias":"baidu/ERNIE-4.5-VL-28B-A3B-Thinking","modelId":"ernie-4.5-vl-28b-a3b-thinking"},{"alias":"baidu/ERNIE-4.5-VL-424B-A47B-PT","modelId":"ernie-4.5-vl-424b-a47b-pt"},{"alias":"baidu/Qianfan-VL-3B","modelId":"qianfan-vl-3b"},{"alias":"baidu/Qianfan-VL-70B","modelId":"qianfan-vl-70b"},{"alias":"baidu/Qianfan-VL-8B","modelId":"qianfan-vl-8b"},{"alias":"tencent/Hunyuan-0.5B-Instruct","modelId":"hunyuan-0.5b-instruct"},{"alias":"tencent/Hunyuan-1.8B-Instruct","modelId":"hunyuan-1.8b-instruct"},{"alias":"tencent/Hunyuan-4B-Instruct","modelId":"hunyuan-4b-instruct"},{"alias":"tencent/Hunyuan-7B-Instruct","modelId":"hunyuan-7b-instruct"},{"alias":"tencent/Hunyuan-7B-Instruct-0124","modelId":"hunyuan-7b-instruct-0124"},{"alias":"tencent/Hunyuan-A13B-Instruct","modelId":"hunyuan-a13b-instruct"},{"alias":"tencent/Hy3","modelId":"hy3"},{"alias":"tencent/Hy3-preview","modelId":"hy3-preview"},{"alias":"tencent/Hy4-preview","modelId":"hy4-preview"},{"alias":"tencent/Tencent-Hunyuan-Large/Hunyuan-A52B-Instruct","modelId":"hunyuan-a52b-instruct"},{"alias":"tencent/WeDLM-7B-Instruct","modelId":"wedlm-7b-instruct"},{"alias":"tencent/WeDLM-8B-Instruct","modelId":"wedlm-8b-instruct"},{"alias":"tencent/Youtu-LLM-2B","modelId":"youtu-llm-2b"},{"alias":"DeepSeek V4 Flash 0424","modelId":"deepseek-v4-flash-0424"},{"alias":"deepseek-v4-flash-0424","modelId":"deepseek-v4-flash-0424"},{"alias":"Sonnet 5.5","modelId":"claude-sonnet-5-5"},{"alias":"anthropic.claude-sonnet-5-5","modelId":"claude-sonnet-5-5"}],"models":[{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":null,"family":"Claude Fable","id":"claude-fable-5","labId":"anthropic","license":"proprietary","name":"Claude Fable 5","releasedOn":null,"retirementOn":null,"status":"active","url":"https://platform.claude.com/docs/en/about-claude/model-deprecations"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":1000000,"family":"Claude Fable","id":"claude-fable-5-1","labId":"anthropic","license":"proprietary","name":"Claude Fable 5.1","releasedOn":null,"retirementOn":null,"status":"active","url":"https://platform.claude.com/docs/en/about-claude/model-deprecations"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":200000,"family":"Claude Haiku","id":"claude-haiku-4-5-20251001","labId":"anthropic","license":"proprietary","name":"Claude Haiku 4.5","releasedOn":null,"retirementOn":null,"status":"active","url":"https://platform.claude.com/docs/en/about-claude/model-deprecations"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":null,"family":"Claude Opus","id":"claude-opus-4-5-20251101","labId":"anthropic","license":"proprietary","name":"Claude Opus 4.5","releasedOn":null,"retirementOn":null,"status":"active","url":"https://platform.claude.com/docs/en/about-claude/model-deprecations"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":null,"family":"Claude Opus","id":"claude-opus-4-6","labId":"anthropic","license":"proprietary","name":"Claude Opus 4.6","releasedOn":null,"retirementOn":null,"status":"active","url":"https://platform.claude.com/docs/en/about-claude/model-deprecations"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":null,"family":"Claude Opus","id":"claude-opus-4-7","labId":"anthropic","license":"proprietary","name":"Claude Opus 4.7","releasedOn":null,"retirementOn":null,"status":"active","url":"https://platform.claude.com/docs/en/about-claude/model-deprecations"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":null,"family":"Claude Opus","id":"claude-opus-4-8","labId":"anthropic","license":"proprietary","name":"Claude Opus 4.8","releasedOn":null,"retirementOn":null,"status":"active","url":"https://platform.claude.com/docs/en/about-claude/model-deprecations"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":1000000,"family":"Claude Opus","id":"claude-opus-5","labId":"anthropic","license":"proprietary","name":"Claude Opus 5","releasedOn":null,"retirementOn":null,"status":"active","url":"https://platform.claude.com/docs/en/about-claude/model-deprecations"},{"auditedOn":"2026-09-22","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":1000000,"family":"Claude Opus","id":"claude-opus-5-5","inputUsdPerMillion":4,"labId":"anthropic","license":"proprietary","name":"Claude Opus 5.5","outputUsdPerMillion":20,"releasedOn":"2026-09-22","retirementOn":null,"status":"active","url":"https://platform.claude.com/docs/en/models/opus-5-5/overview"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":null,"family":"Claude Sonnet","id":"claude-sonnet-4-5-20250929","labId":"anthropic","license":"proprietary","name":"Claude Sonnet 4.5","releasedOn":null,"retirementOn":null,"status":"active","url":"https://platform.claude.com/docs/en/about-claude/model-deprecations"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":null,"family":"Claude Sonnet","id":"claude-sonnet-4-6","labId":"anthropic","license":"proprietary","name":"Claude Sonnet 4.6","releasedOn":null,"retirementOn":null,"status":"active","url":"https://platform.claude.com/docs/en/about-claude/model-deprecations"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":1000000,"family":"Claude Sonnet","id":"claude-sonnet-5","labId":"anthropic","license":"proprietary","name":"Claude Sonnet 5","releasedOn":null,"retirementOn":null,"status":"active","url":"https://platform.claude.com/docs/en/about-claude/model-deprecations"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"DeepSeek R1","id":"deepseek-r1-0528","labId":"deepseek","license":"mit","name":"DeepSeek-R1-0528","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/deepseek-ai/DeepSeek-R1-0528"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"DeepSeek R1","id":"deepseek-r1-0528-qwen3-8b","labId":"deepseek","license":"mit","name":"DeepSeek-R1-0528-Qwen3-8B","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/deepseek-ai/DeepSeek-R1-0528-Qwen3-8B"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"DeepSeek R1","id":"deepseek-r1-distill-llama-70b","labId":"deepseek","license":"mit","name":"DeepSeek-R1-Distill-Llama-70B","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/deepseek-ai/DeepSeek-R1-Distill-Llama-70B"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"DeepSeek R1","id":"deepseek-r1-distill-llama-8b","labId":"deepseek","license":"mit","name":"DeepSeek-R1-Distill-Llama-8B","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/deepseek-ai/DeepSeek-R1-Distill-Llama-8B"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"DeepSeek R1","id":"deepseek-r1-distill-qwen-1.5b","labId":"deepseek","license":"mit","name":"DeepSeek-R1-Distill-Qwen-1.5B","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/deepseek-ai/DeepSeek-R1-Distill-Qwen-1.5B"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"DeepSeek R1","id":"deepseek-r1-distill-qwen-14b","labId":"deepseek","license":"mit","name":"DeepSeek-R1-Distill-Qwen-14B","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/deepseek-ai/DeepSeek-R1-Distill-Qwen-14B"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"DeepSeek R1","id":"deepseek-r1-distill-qwen-32b","labId":"deepseek","license":"mit","name":"DeepSeek-R1-Distill-Qwen-32B","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/deepseek-ai/DeepSeek-R1-Distill-Qwen-32B"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"DeepSeek R1","id":"deepseek-r1-distill-qwen-7b","labId":"deepseek","license":"mit","name":"DeepSeek-R1-Distill-Qwen-7B","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/deepseek-ai/DeepSeek-R1-Distill-Qwen-7B"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"DeepSeek V3","id":"deepseek-v3.2","labId":"deepseek","license":"mit","name":"DeepSeek-V3.2","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/deepseek-ai/DeepSeek-V3.2"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"DeepSeek V3","id":"deepseek-v3.2-speciale","labId":"deepseek","license":"mit","name":"DeepSeek-V3.2-Speciale","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/deepseek-ai/DeepSeek-V3.2-Speciale"},{"auditedOn":"2026-09-06","availability":"Documented provider API and open weights for self-hosting","contextTokens":1000000,"family":"DeepSeek V4","id":"deepseek-v4-flash","labId":"deepseek","license":"mit","name":"DeepSeek-V4-Flash-0731","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash-0731"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":1000000,"family":"DeepSeek V4","id":"deepseek-v4-flash-vision-exp","labId":"deepseek","license":"mit","name":"DeepSeek-V4-Flash-Vision-Exp","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash-Vision-Exp"},{"auditedOn":"2026-09-06","availability":"Documented provider API and open weights for self-hosting","contextTokens":1000000,"family":"DeepSeek V4","id":"deepseek-v4-pro","labId":"deepseek","license":"mit","name":"DeepSeek V4-Pro 0813","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813"},{"auditedOn":"2026-09-06","availability":"Open instruction-tuned weights for coding","contextTokens":null,"family":"CodeGemma","id":"codegemma-7b-it","labId":"google","license":null,"name":"CodeGemma 7B IT","releasedOn":null,"retirementOn":null,"status":"active","url":"https://ai.google.dev/gemma/docs/codegemma"},{"auditedOn":"2026-09-06","availability":"Experimental open weights; text output from text, image and video","contextTokens":null,"family":"DiffusionGemma","id":"diffusiongemma-26b-a4b","labId":"google","license":null,"name":"DiffusionGemma 26B A4B","releasedOn":null,"retirementOn":null,"status":"experimental","url":"https://ai.google.dev/gemma/docs/diffusiongemma"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":null,"family":"Gemini 2.5","id":"gemini-2.5-flash","labId":"google","license":"proprietary","name":"Gemini 2.5 Flash","releasedOn":null,"retirementOn":null,"status":"active","url":"https://ai.google.dev/gemini-api/docs/models"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":null,"family":"Gemini 2.5","id":"gemini-2.5-flash-lite","labId":"google","license":"proprietary","name":"Gemini 2.5 Flash-Lite","releasedOn":null,"retirementOn":null,"status":"active","url":"https://ai.google.dev/gemini-api/docs/models"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":null,"family":"Gemini 2.5","id":"gemini-2.5-pro","labId":"google","license":"proprietary","name":"Gemini 2.5 Pro","releasedOn":null,"retirementOn":null,"status":"active","url":"https://ai.google.dev/gemini-api/docs/models"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":null,"family":"Gemini 3","id":"gemini-3-flash-preview","labId":"google","license":"proprietary","name":"Gemini 3 Flash Preview","releasedOn":null,"retirementOn":null,"status":"preview","url":"https://ai.google.dev/gemini-api/docs/models"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":null,"family":"Gemini 3","id":"gemini-3.1-flash-lite","labId":"google","license":"proprietary","name":"Gemini 3.1 Flash-Lite","releasedOn":null,"retirementOn":null,"status":"active","url":"https://ai.google.dev/gemini-api/docs/models"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":null,"family":"Gemini 3","id":"gemini-3.1-pro","labId":"google","license":"proprietary","name":"Gemini 3.1 Pro","releasedOn":null,"retirementOn":null,"status":"preview","url":"https://ai.google.dev/gemini-api/docs/models"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":null,"family":"Gemini 3","id":"gemini-3.5-flash","labId":"google","license":"proprietary","name":"Gemini 3.5 Flash","releasedOn":null,"retirementOn":null,"status":"active","url":"https://ai.google.dev/gemini-api/docs/models"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":null,"family":"Gemini 3","id":"gemini-3.5-flash-lite","labId":"google","license":"proprietary","name":"Gemini 3.5 Flash-Lite","releasedOn":null,"retirementOn":null,"status":"active","url":"https://ai.google.dev/gemini-api/docs/models"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":null,"family":"Gemini 3","id":"gemini-3.6-flash","labId":"google","license":"proprietary","name":"Gemini 3.6 Flash","releasedOn":null,"retirementOn":null,"status":"active","url":"https://ai.google.dev/gemini-api/docs/models"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":null,"family":"Gemini 3","id":"gemini-3.7-flash","labId":"google","license":"proprietary","name":"Gemini 3.7 Flash","releasedOn":null,"retirementOn":null,"status":"active","url":"https://ai.google.dev/gemini-api/docs/models"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":null,"family":"Gemini 3","id":"gemini-3.8-flash","labId":"google","license":"proprietary","name":"Gemini 3.8 Flash","releasedOn":null,"retirementOn":null,"status":"active","url":"https://ai.google.dev/gemini-api/docs/models"},{"auditedOn":"2026-09-06","availability":"Open instruction-tuned weights available for self-hosting","contextTokens":null,"family":"Gemma 1","id":"gemma-1-2b-it","labId":"google","license":null,"name":"Gemma 1 2B IT","releasedOn":null,"retirementOn":null,"status":"active","url":"https://ai.google.dev/gemma/docs/core/model_card"},{"auditedOn":"2026-09-06","availability":"Open instruction-tuned weights available for self-hosting","contextTokens":null,"family":"Gemma 1","id":"gemma-1-7b-it","labId":"google","license":null,"name":"Gemma 1 7B IT","releasedOn":null,"retirementOn":null,"status":"active","url":"https://ai.google.dev/gemma/docs/core/model_card"},{"auditedOn":"2026-09-06","availability":"Open instruction-tuned weights available for self-hosting","contextTokens":null,"family":"Gemma 1.1","id":"gemma-1.1-2b-it","labId":"google","license":null,"name":"Gemma 1.1 2B IT","releasedOn":null,"retirementOn":null,"status":"active","url":"https://ai.google.dev/gemma/docs/core/model_card"},{"auditedOn":"2026-09-06","availability":"Open instruction-tuned weights available for self-hosting","contextTokens":null,"family":"Gemma 1.1","id":"gemma-1.1-7b-it","labId":"google","license":null,"name":"Gemma 1.1 7B IT","releasedOn":null,"retirementOn":null,"status":"active","url":"https://ai.google.dev/gemma/docs/core/model_card"},{"auditedOn":"2026-09-06","availability":"Open instruction-tuned weights available for self-hosting","contextTokens":null,"family":"Gemma 2","id":"gemma-2-27b-it","labId":"google","license":null,"name":"Gemma 2 27B IT","releasedOn":null,"retirementOn":null,"status":"active","url":"https://ai.google.dev/gemma/docs/core/model_card_2"},{"auditedOn":"2026-09-06","availability":"Open instruction-tuned weights available for self-hosting","contextTokens":null,"family":"Gemma 2","id":"gemma-2-2b-it","labId":"google","license":null,"name":"Gemma 2 2B IT","releasedOn":null,"retirementOn":null,"status":"active","url":"https://ai.google.dev/gemma/docs/core/model_card_2"},{"auditedOn":"2026-09-06","availability":"Open instruction-tuned weights available for self-hosting","contextTokens":null,"family":"Gemma 2","id":"gemma-2-9b-it","labId":"google","license":null,"name":"Gemma 2 9B IT","releasedOn":null,"retirementOn":null,"status":"active","url":"https://ai.google.dev/gemma/docs/core/model_card_2"},{"auditedOn":"2026-09-06","availability":"Open instruction-tuned weights available for self-hosting","contextTokens":131072,"family":"Gemma 3","id":"gemma-3-12b-it","labId":"google","license":null,"name":"Gemma 3 12B IT","releasedOn":null,"retirementOn":null,"status":"active","url":"https://ai.google.dev/gemma/docs/core/model_card_3"},{"auditedOn":"2026-09-06","availability":"Open instruction-tuned weights available for self-hosting","contextTokens":32768,"family":"Gemma 3","id":"gemma-3-1b-it","labId":"google","license":null,"name":"Gemma 3 1B IT","releasedOn":null,"retirementOn":null,"status":"active","url":"https://ai.google.dev/gemma/docs/core/model_card_3"},{"auditedOn":"2026-09-06","availability":"Open instruction-tuned weights available for self-hosting","contextTokens":32768,"family":"Gemma 3","id":"gemma-3-270m-it","labId":"google","license":null,"name":"Gemma 3 270M IT","releasedOn":null,"retirementOn":null,"status":"active","url":"https://ai.google.dev/gemma/docs/core/model_card_3"},{"auditedOn":"2026-09-06","availability":"Open instruction-tuned weights available for self-hosting","contextTokens":131072,"family":"Gemma 3","id":"gemma-3-27b-it","labId":"google","license":null,"name":"Gemma 3 27B IT","releasedOn":null,"retirementOn":null,"status":"active","url":"https://ai.google.dev/gemma/docs/core/model_card_3"},{"auditedOn":"2026-09-06","availability":"Open instruction-tuned weights available for self-hosting","contextTokens":131072,"family":"Gemma 3","id":"gemma-3-4b-it","labId":"google","license":null,"name":"Gemma 3 4B IT","releasedOn":null,"retirementOn":null,"status":"active","url":"https://ai.google.dev/gemma/docs/core/model_card_3"},{"auditedOn":"2026-09-06","availability":"Open instruction-tuned weights available for self-hosting","contextTokens":32768,"family":"Gemma 3n","id":"gemma-3n-e2b-it","labId":"google","license":null,"name":"Gemma 3n E2B IT","releasedOn":null,"retirementOn":null,"status":"active","url":"https://ai.google.dev/gemma/docs/gemma-3n"},{"auditedOn":"2026-09-06","availability":"Open instruction-tuned weights available for self-hosting","contextTokens":32768,"family":"Gemma 3n","id":"gemma-3n-e4b-it","labId":"google","license":null,"name":"Gemma 3n E4B IT","releasedOn":null,"retirementOn":null,"status":"active","url":"https://ai.google.dev/gemma/docs/gemma-3n"},{"auditedOn":"2026-09-06","availability":"Open instruction-tuned weights available for self-hosting","contextTokens":262144,"family":"Gemma 4","id":"gemma-4-12b-it","labId":"google","license":"Apache 2.0","name":"Gemma 4 12B IT","releasedOn":null,"retirementOn":null,"status":"active","url":"https://ai.google.dev/gemma/docs/core/model_card_4"},{"auditedOn":"2026-09-06","availability":"Open instruction-tuned weights available for self-hosting","contextTokens":262144,"family":"Gemma 4","id":"gemma-4-26b-a4b-it","labId":"google","license":"Apache 2.0","name":"Gemma 4 26B-A4B IT","releasedOn":null,"retirementOn":null,"status":"active","url":"https://ai.google.dev/gemma/docs/core/model_card_4"},{"auditedOn":"2026-09-06","availability":"Open instruction-tuned weights available for self-hosting","contextTokens":262144,"family":"Gemma 4","id":"gemma-4-31b-it","labId":"google","license":"Apache 2.0","name":"Gemma 4 31B IT","releasedOn":null,"retirementOn":null,"status":"active","url":"https://ai.google.dev/gemma/docs/core/model_card_4"},{"auditedOn":"2026-09-06","availability":"Open instruction-tuned weights available for self-hosting","contextTokens":131072,"family":"Gemma 4","id":"gemma-4-e2b-it","labId":"google","license":"Apache 2.0","name":"Gemma 4 E2B IT","releasedOn":null,"retirementOn":null,"status":"active","url":"https://ai.google.dev/gemma/docs/core/model_card_4"},{"auditedOn":"2026-09-06","availability":"Open instruction-tuned weights available for self-hosting","contextTokens":131072,"family":"Gemma 4","id":"gemma-4-e4b-it","labId":"google","license":"Apache 2.0","name":"Gemma 4 E4B IT","releasedOn":null,"retirementOn":null,"status":"active","url":"https://ai.google.dev/gemma/docs/core/model_card_4"},{"auditedOn":"2026-09-06","availability":"Open instruction-tuned weights for self-hosting","contextTokens":null,"family":"RecurrentGemma","id":"recurrentgemma-2b-it","labId":"google","license":null,"name":"RecurrentGemma 2B IT","releasedOn":null,"retirementOn":null,"status":"active","url":"https://ai.google.dev/gemma/docs/recurrentgemma/model_card"},{"auditedOn":"2026-09-06","availability":"Open instruction-tuned weights for self-hosting","contextTokens":null,"family":"RecurrentGemma","id":"recurrentgemma-9b-it","labId":"google","license":null,"name":"RecurrentGemma 9B IT","releasedOn":null,"retirementOn":null,"status":"active","url":"https://ai.google.dev/gemma/docs/recurrentgemma/model_card"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Llama 3.1","id":"llama-3.1-405b-instruct","labId":"meta","license":"llama3.1","name":"Llama-3.1-405B-Instruct","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/meta-llama/Llama-3.1-405B-Instruct"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Llama 3.1","id":"llama-3.1-70b-instruct","labId":"meta","license":"llama3.1","name":"Llama-3.1-70B-Instruct","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/meta-llama/Llama-3.1-70B-Instruct"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Llama 3.1","id":"llama-3.1-8b-instruct","labId":"meta","license":"llama3.1","name":"Llama-3.1-8B-Instruct","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/meta-llama/Llama-3.1-8B-Instruct"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Llama 3.2","id":"llama-3.2-11b-vision-instruct","labId":"meta","license":"llama3.2","name":"Llama-3.2-11B-Vision-Instruct","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/meta-llama/Llama-3.2-11B-Vision-Instruct"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Llama 3.2","id":"llama-3.2-1b-instruct","labId":"meta","license":"llama3.2","name":"Llama-3.2-1B-Instruct","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Llama 3.2","id":"llama-3.2-3b-instruct","labId":"meta","license":"llama3.2","name":"Llama-3.2-3B-Instruct","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/meta-llama/Llama-3.2-3B-Instruct"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Llama 3.2","id":"llama-3.2-90b-vision-instruct","labId":"meta","license":"llama3.2","name":"Llama-3.2-90B-Vision-Instruct","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/meta-llama/Llama-3.2-90B-Vision-Instruct"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Llama 3.3","id":"llama-3.3-70b-instruct","labId":"meta","license":"llama3.3","name":"Llama-3.3-70B-Instruct","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/meta-llama/Llama-3.3-70B-Instruct"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Llama 4","id":"llama-4-maverick","labId":"meta","license":null,"name":"Llama 4 Maverick","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/meta-llama/Llama-4-Maverick-17B-128E-Instruct"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Llama 4","id":"llama-4-scout","labId":"meta","license":null,"name":"Llama-4-Scout-17B-16E-Instruct","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/meta-llama/Llama-4-Scout-17B-16E-Instruct"},{"auditedOn":"2026-09-06","availability":"Downloadable official model weights","contextTokens":131072,"family":"Muse Glimmer","id":"muse-glimmer-30b","labId":"meta","license":"Apache-2.0","name":"Muse Glimmer 30B","releasedOn":"2026-08-10","retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/meta-models/Muse-Glimmer-30B"},{"auditedOn":"2026-09-06","availability":"Documented Meta Model API; public preview at release; current version-specific serving requires authenticated verification","contextTokens":1000000,"family":"Muse Spark","id":"muse-spark-1.1","labId":"meta","license":"proprietary","name":"Muse Spark 1.1","releasedOn":"2026-07-09","retirementOn":null,"status":"availability-unverified","url":"https://research.meta.ai/blog/introducing-muse-spark-meta-model-api"},{"auditedOn":"2026-09-06","availability":"Documented Meta Model API; current version-specific serving requires authenticated verification","contextTokens":null,"family":"Muse Spark","id":"muse-spark-1.2","labId":"meta","license":"proprietary","name":"Muse Spark 1.2","releasedOn":"2026-08-05","retirementOn":null,"status":"availability-unverified","url":"https://research.meta.ai/blog/introducing-muse-code-and-muse-spark-1-2"},{"auditedOn":"2026-09-06","availability":"Documented Meta Model API and Muse Code","contextTokens":null,"family":"Muse Spark","id":"muse-spark-1.3","labId":"meta","license":"proprietary","name":"Muse Spark 1.3","releasedOn":"2026-09-02","retirementOn":null,"status":"active","url":"https://research.meta.ai/blog/introducing-muse-spark-1-3"},{"auditedOn":"2026-09-06","availability":"Documented provider API; account and region restrictions may apply","contextTokens":null,"family":"Codestral","id":"codestral-2508","labId":"mistral","license":"proprietary","name":"Codestral 2508","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://docs.mistral.ai/models"},{"auditedOn":"2026-09-06","availability":"Documented provider API; deprecated and still accessible until retirement","contextTokens":256000,"family":"Devstral","id":"devstral-2512","labId":"mistral","license":"Modified MIT","name":"Devstral 2","releasedOn":"2025-12-09","retirementOn":null,"status":"deprecated","url":"https://docs.mistral.ai/models/devstral-2-25-12"},{"auditedOn":"2026-09-06","availability":"Documented provider API; deprecated and still accessible until retirement","contextTokens":128000,"family":"Devstral","id":"devstral-medium-2507","labId":"mistral","license":"proprietary","name":"Devstral Medium 1.0","releasedOn":"2025-07-10","retirementOn":null,"status":"deprecated","url":"https://docs.mistral.ai/models/devstral-medium-1-0-25-07"},{"auditedOn":"2026-09-06","availability":"Documented provider API; deprecated and still accessible until retirement","contextTokens":128000,"family":"Devstral","id":"devstral-small-2507","labId":"mistral","license":"Apache 2.0","name":"Devstral Small 1.1","releasedOn":"2025-07-10","retirementOn":null,"status":"deprecated","url":"https://docs.mistral.ai/models/devstral-small-1-1-25-07"},{"auditedOn":"2026-09-06","availability":"Documented provider API; deprecated and still accessible until retirement","contextTokens":256000,"family":"Devstral","id":"labs-devstral-small-2512","labId":"mistral","license":"Apache 2.0","name":"Devstral Small 2","releasedOn":"2025-12-09","retirementOn":null,"status":"deprecated","url":"https://docs.mistral.ai/models/devstral-small-2-25-12"},{"auditedOn":"2026-09-06","availability":"Documented provider API; deprecated and still accessible until retirement","contextTokens":32000,"family":"Mistral","id":"labs-mistral-small-creative","labId":"mistral","license":null,"name":"Mistral Small Creative","releasedOn":"2025-12-11","retirementOn":null,"status":"deprecated","url":"https://docs.mistral.ai/models/mistral-small-creative-25-12"},{"auditedOn":"2026-09-06","availability":"Documented provider API; account and region restrictions may apply","contextTokens":256000,"family":"Leanstral","id":"leanstral-1.5","labId":"mistral","license":"Apache 2.0","name":"Leanstral 1.5","releasedOn":"2026-06-30","retirementOn":null,"status":"api-listed","url":"https://docs.mistral.ai/models/leanstral-1-5"},{"auditedOn":"2026-09-06","availability":"Documented provider API; deprecated and still accessible until retirement","contextTokens":128000,"family":"Magistral","id":"magistral-medium-2509","labId":"mistral","license":"proprietary","name":"Magistral Medium 1.2","releasedOn":"2025-09-18","retirementOn":null,"status":"deprecated","url":"https://docs.mistral.ai/models/magistral-medium-1-2-25-09"},{"auditedOn":"2026-09-06","availability":"Documented provider API; deprecated and still accessible until retirement","contextTokens":128000,"family":"Magistral","id":"magistral-small-2509","labId":"mistral","license":"Apache 2.0","name":"Magistral Small 1.2","releasedOn":"2025-09-18","retirementOn":null,"status":"deprecated","url":"https://docs.mistral.ai/models/magistral-small-1-2-25-09"},{"auditedOn":"2026-09-06","availability":"Documented provider API; account and region restrictions may apply","contextTokens":null,"family":"Ministral","id":"ministral-3-14b","labId":"mistral","license":"Apache 2.0","name":"Ministral 3 14B","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://docs.mistral.ai/models"},{"auditedOn":"2026-09-06","availability":"Documented provider API; account and region restrictions may apply","contextTokens":null,"family":"Ministral","id":"ministral-3-3b","labId":"mistral","license":"Apache 2.0","name":"Ministral 3 3B","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://docs.mistral.ai/models"},{"auditedOn":"2026-09-06","availability":"Documented provider API; account and region restrictions may apply","contextTokens":null,"family":"Ministral","id":"ministral-3-8b","labId":"mistral","license":"Apache 2.0","name":"Ministral 3 8B","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://docs.mistral.ai/models"},{"auditedOn":"2026-09-06","availability":"Documented provider API; deprecated and still accessible until retirement","contextTokens":128000,"family":"Ministral","id":"ministral-3b-2410","labId":"mistral","license":"proprietary","name":"Ministral 3B","releasedOn":"2024-10-09","retirementOn":null,"status":"deprecated","url":"https://docs.mistral.ai/models/ministral-3b-24-1"},{"auditedOn":"2026-09-06","availability":"Documented provider API; deprecated and still accessible until retirement","contextTokens":128000,"family":"Ministral","id":"ministral-8b-2410","labId":"mistral","license":"MRL","name":"Ministral 8B","releasedOn":"2024-10-09","retirementOn":null,"status":"deprecated","url":"https://docs.mistral.ai/models/ministral-8b-24-1"},{"auditedOn":"2026-09-06","availability":"Documented provider API; deprecated and still accessible until retirement","contextTokens":128000,"family":"Mistral","id":"mistral-large-2411","labId":"mistral","license":"MRL","name":"Mistral Large 2.1","releasedOn":"2024-11-18","retirementOn":null,"status":"deprecated","url":"https://docs.mistral.ai/models/mistral-large-2-1-24-11"},{"auditedOn":"2026-09-06","availability":"Documented provider API; account and region restrictions may apply","contextTokens":null,"family":"Mistral","id":"mistral-large-3","labId":"mistral","license":"Apache 2.0","name":"Mistral Large 3","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://docs.mistral.ai/models"},{"auditedOn":"2026-09-06","availability":"Documented provider API; deprecated and still accessible until retirement","contextTokens":128000,"family":"Mistral","id":"mistral-medium-2505","labId":"mistral","license":"proprietary","name":"Mistral Medium 3","releasedOn":"2025-05-07","retirementOn":null,"status":"deprecated","url":"https://docs.mistral.ai/models/mistral-medium-3-25-05"},{"auditedOn":"2026-09-06","availability":"Documented provider API; deprecated and still accessible until retirement","contextTokens":128000,"family":"Mistral","id":"mistral-medium-2508","labId":"mistral","license":"proprietary","name":"Mistral Medium 3.1","releasedOn":"2025-08-12","retirementOn":null,"status":"deprecated","url":"https://docs.mistral.ai/models/mistral-medium-3-1-25-08"},{"auditedOn":"2026-09-06","availability":"Documented provider API; account and region restrictions may apply","contextTokens":256000,"family":"Mistral","id":"mistral-medium-3.5","labId":"mistral","license":"Modified MIT","name":"Mistral Medium 3.5","releasedOn":"2026-04-28","retirementOn":null,"status":"api-listed","url":"https://docs.mistral.ai/models/mistral-medium-3-5-26-04"},{"auditedOn":"2026-09-06","availability":"Documented provider API; deprecated and still accessible until retirement","contextTokens":128000,"family":"Mistral","id":"mistral-small-2506","labId":"mistral","license":"Apache 2.0","name":"Mistral Small 3.2","releasedOn":"2025-06-20","retirementOn":null,"status":"deprecated","url":"https://docs.mistral.ai/models/mistral-small-3-2-25-06"},{"auditedOn":"2026-09-06","availability":"Documented provider API; account and region restrictions may apply","contextTokens":256000,"family":"Mistral","id":"mistral-small-4","labId":"mistral","license":"Apache 2.0","name":"Mistral Small 4","releasedOn":"2026-03-16","retirementOn":null,"status":"api-listed","url":"https://docs.mistral.ai/models/mistral-small-4-0-26-03"},{"auditedOn":"2026-09-06","availability":"Documented provider API; deprecated and still accessible until retirement","contextTokens":128000,"family":"Mistral","id":"open-mistral-nemo-2407","labId":"mistral","license":"Apache 2.0","name":"Mistral Nemo 12B","releasedOn":"2024-07-18","retirementOn":null,"status":"deprecated","url":"https://docs.mistral.ai/models/mistral-nemo-12b-24-07"},{"auditedOn":"2026-09-06","availability":"Documented provider API; deprecated and still accessible until retirement","contextTokens":128000,"family":"Pixtral","id":"pixtral-12b-2409","labId":"mistral","license":"Apache 2.0","name":"Pixtral 12B","releasedOn":"2024-09-11","retirementOn":null,"status":"deprecated","url":"https://docs.mistral.ai/models/pixtral-12b-24-09"},{"auditedOn":"2026-09-06","availability":"Documented provider API; deprecated and still accessible until retirement","contextTokens":128000,"family":"Pixtral","id":"pixtral-large-2411","labId":"mistral","license":"MRL","name":"Pixtral Large","releasedOn":"2024-11-18","retirementOn":null,"status":"deprecated","url":"https://docs.mistral.ai/models/pixtral-large-24-11"},{"auditedOn":"2026-09-06","availability":"Documented provider API; deprecated and still accessible until retirement","contextTokens":32000,"family":"Voxtral","id":"voxtral-mini-2507","labId":"mistral","license":"Apache 2.0","name":"Voxtral Mini","releasedOn":"2025-07-15","retirementOn":null,"status":"deprecated","url":"https://docs.mistral.ai/models/voxtral-mini-25-07"},{"auditedOn":"2026-09-06","availability":"Documented provider API; account and region restrictions may apply","contextTokens":null,"family":"Voxtral","id":"voxtral-small","labId":"mistral","license":"Apache 2.0","name":"Voxtral Small","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://docs.mistral.ai/models"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Kimi","id":"kimi-dev-72b","labId":"moonshot","license":"mit","name":"Kimi-Dev-72B","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/moonshotai/Kimi-Dev-72B"},{"auditedOn":"2026-09-06","availability":"Documented provider API and open weights for self-hosting","contextTokens":256000,"family":"Kimi","id":"kimi-k2.6","labId":"moonshot","license":null,"name":"Kimi-K2.6","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://huggingface.co/moonshotai/Kimi-K2.6"},{"auditedOn":"2026-09-06","availability":"Documented provider API and open weights for self-hosting","contextTokens":256000,"family":"Kimi","id":"kimi-k2.7-code","labId":"moonshot","license":null,"name":"Kimi-K2.7-Code","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://huggingface.co/moonshotai/Kimi-K2.7-Code"},{"auditedOn":"2026-09-06","availability":"Documented provider API and open weights for self-hosting","contextTokens":1000000,"family":"Kimi","id":"kimi-k3","labId":"moonshot","license":null,"name":"Kimi K3","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://huggingface.co/moonshotai/Kimi-K3"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Kimi","id":"kimi-linear-48b-a3b-instruct","labId":"moonshot","license":"mit","name":"Kimi-Linear-48B-A3B-Instruct","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/moonshotai/Kimi-Linear-48B-A3B-Instruct"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Kimi","id":"kimi-vl-a3b-instruct","labId":"moonshot","license":"mit","name":"Kimi-VL-A3B-Instruct","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/moonshotai/Kimi-VL-A3B-Instruct"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Kimi","id":"kimi-vl-a3b-thinking-2506","labId":"moonshot","license":"mit","name":"Kimi-VL-A3B-Thinking-2506","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/moonshotai/Kimi-VL-A3B-Thinking-2506"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Nemotron 3","id":"nemotron-3-nano-omni-30b-a3b-reasoning-bf16","labId":"nvidia","license":null,"name":"Nemotron-3-Nano-Omni-30B-A3B-Reasoning-BF16","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/nvidia/Nemotron-3-Nano-Omni-30B-A3B-Reasoning-BF16"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Nemotron 3","id":"nemotron-3-ultra","labId":"nvidia","license":null,"name":"NVIDIA Nemotron 3 Ultra","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Nemotron 3","id":"nvidia-nemotron-3-nano-30b-a3b-bf16","labId":"nvidia","license":null,"name":"NVIDIA-Nemotron-3-Nano-30B-A3B-BF16","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-BF16"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Nemotron 3","id":"nvidia-nemotron-3-nano-4b-bf16","labId":"nvidia","license":null,"name":"NVIDIA-Nemotron-3-Nano-4B-BF16","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Nano-4B-BF16"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Nemotron 3","id":"nvidia-nemotron-3-super-120b-a12b-bf16","labId":"nvidia","license":null,"name":"NVIDIA-Nemotron-3-Super-120B-A12B-BF16","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Nemotron 3.5","id":"nvidia-nemotron-3.5-lightning-30b-a3b-bf16","labId":"nvidia","license":null,"name":"NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Nemotron 2","id":"nvidia-nemotron-nano-12b-v2","labId":"nvidia","license":null,"name":"NVIDIA-Nemotron-Nano-12B-v2","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/nvidia/NVIDIA-Nemotron-Nano-12B-v2"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Nemotron 2","id":"nvidia-nemotron-nano-12b-v2-vl-bf16","labId":"nvidia","license":null,"name":"NVIDIA-Nemotron-Nano-12B-v2-VL-BF16","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/nvidia/NVIDIA-Nemotron-Nano-12B-v2-VL-BF16"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Nemotron 2","id":"nvidia-nemotron-nano-9b-v2","labId":"nvidia","license":null,"name":"NVIDIA-Nemotron-Nano-9B-v2","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/nvidia/NVIDIA-Nemotron-Nano-9B-v2"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Nemotron 2","id":"nvidia-nemotron-nano-9b-v2-japanese","labId":"nvidia","license":null,"name":"NVIDIA-Nemotron-Nano-9B-v2-Japanese","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/nvidia/NVIDIA-Nemotron-Nano-9B-v2-Japanese"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":null,"family":"gpt-3.5","id":"gpt-3.5-turbo","labId":"openai","license":"proprietary","name":"GPT-3.5-turbo","releasedOn":null,"retirementOn":"2026-10-23","status":"deprecated","url":"https://developers.openai.com/api/docs/deprecations"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":null,"family":"gpt-3.5","id":"gpt-3.5-turbo-1106","labId":"openai","license":"proprietary","name":"GPT-3.5-turbo-1106","releasedOn":null,"retirementOn":"2026-09-28","status":"deprecated","url":"https://developers.openai.com/api/docs/deprecations"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":null,"family":"gpt-3.5","id":"gpt-3.5-turbo-instruct","labId":"openai","license":"proprietary","name":"GPT-3.5-turbo-instruct","releasedOn":null,"retirementOn":"2026-09-28","status":"deprecated","url":"https://developers.openai.com/api/docs/deprecations"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":null,"family":"gpt-4","id":"gpt-4","labId":"openai","license":"proprietary","name":"GPT-4","releasedOn":null,"retirementOn":"2026-10-23","status":"deprecated","url":"https://developers.openai.com/api/docs/deprecations"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":null,"family":"gpt-4","id":"gpt-4-turbo","labId":"openai","license":"proprietary","name":"GPT-4-turbo","releasedOn":null,"retirementOn":"2026-10-23","status":"deprecated","url":"https://developers.openai.com/api/docs/deprecations"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":null,"family":"gpt-4.1","id":"gpt-4.1","labId":"openai","license":"proprietary","name":"GPT-4.1","releasedOn":null,"retirementOn":null,"status":"active","url":"https://developers.openai.com/api/docs/models/all"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":null,"family":"gpt-4.1","id":"gpt-4.1-mini","labId":"openai","license":"proprietary","name":"GPT-4.1 Mini","releasedOn":null,"retirementOn":null,"status":"active","url":"https://developers.openai.com/api/docs/models/all"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":null,"family":"gpt-4.1","id":"gpt-4.1-nano","labId":"openai","license":"proprietary","name":"GPT-4.1-nano","releasedOn":null,"retirementOn":"2026-10-23","status":"deprecated","url":"https://developers.openai.com/api/docs/deprecations"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":null,"family":"gpt-4o","id":"gpt-4o","labId":"openai","license":"proprietary","name":"GPT-4o","releasedOn":null,"retirementOn":null,"status":"active","url":"https://developers.openai.com/api/docs/models/all"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":null,"family":"gpt-4o","id":"gpt-4o-mini","labId":"openai","license":"proprietary","name":"GPT-4o Mini","releasedOn":null,"retirementOn":null,"status":"active","url":"https://developers.openai.com/api/docs/models/all"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":null,"family":"gpt-5","id":"gpt-5","labId":"openai","license":"proprietary","name":"GPT-5","releasedOn":null,"retirementOn":"2026-12-11","status":"deprecated","url":"https://developers.openai.com/api/docs/models/all"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":null,"family":"gpt-5","id":"gpt-5-mini","labId":"openai","license":"proprietary","name":"GPT-5 Mini","releasedOn":null,"retirementOn":"2026-12-11","status":"deprecated","url":"https://developers.openai.com/api/docs/models/all"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":null,"family":"gpt-5","id":"gpt-5-nano","labId":"openai","license":"proprietary","name":"GPT-5 Nano","releasedOn":null,"retirementOn":"2026-12-11","status":"deprecated","url":"https://developers.openai.com/api/docs/models/all"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":null,"family":"gpt-5","id":"gpt-5-pro","labId":"openai","license":"proprietary","name":"GPT-5 Pro","releasedOn":null,"retirementOn":"2026-12-11","status":"deprecated","url":"https://developers.openai.com/api/docs/models/all"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":null,"family":"gpt-5.1","id":"gpt-5.1","labId":"openai","license":"proprietary","name":"GPT-5.1","releasedOn":null,"retirementOn":null,"status":"active","url":"https://developers.openai.com/api/docs/models/all"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":null,"family":"gpt-5.2","id":"gpt-5.2","labId":"openai","license":"proprietary","name":"GPT-5.2","releasedOn":null,"retirementOn":null,"status":"active","url":"https://developers.openai.com/api/docs/models/all"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":null,"family":"gpt-5.2","id":"gpt-5.2-pro","labId":"openai","license":"proprietary","name":"GPT-5.2 Pro","releasedOn":null,"retirementOn":null,"status":"active","url":"https://developers.openai.com/api/docs/models/all"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":null,"family":"gpt-5.3","id":"gpt-5.3-codex","labId":"openai","license":"proprietary","name":"GPT-5.3 Codex","releasedOn":null,"retirementOn":null,"status":"active","url":"https://developers.openai.com/api/docs/models/all"},{"auditedOn":"2026-09-06","availability":"Text-only research preview for ChatGPT Pro users in Codex","contextTokens":null,"family":"GPT-5.3","id":"gpt-5.3-codex-spark","labId":"openai","license":"proprietary","name":"GPT-5.3 Codex Spark","releasedOn":null,"retirementOn":null,"status":"preview","url":"https://learn.chatgpt.com/docs/models"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":null,"family":"gpt-5.4","id":"gpt-5.4","labId":"openai","license":"proprietary","name":"GPT-5.4","releasedOn":null,"retirementOn":null,"status":"active","url":"https://developers.openai.com/api/docs/models/all"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":null,"family":"gpt-5.4","id":"gpt-5.4-mini","labId":"openai","license":"proprietary","name":"GPT-5.4 Mini","releasedOn":null,"retirementOn":null,"status":"active","url":"https://developers.openai.com/api/docs/models/all"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":null,"family":"gpt-5.4","id":"gpt-5.4-nano","labId":"openai","license":"proprietary","name":"GPT-5.4 Nano","releasedOn":null,"retirementOn":null,"status":"active","url":"https://developers.openai.com/api/docs/models/all"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":null,"family":"gpt-5.4","id":"gpt-5.4-pro","labId":"openai","license":"proprietary","name":"GPT-5.4 Pro","releasedOn":null,"retirementOn":null,"status":"active","url":"https://developers.openai.com/api/docs/models/all"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":null,"family":"gpt-5.5","id":"gpt-5.5","labId":"openai","license":"proprietary","name":"GPT-5.5","releasedOn":null,"retirementOn":null,"status":"active","url":"https://developers.openai.com/api/docs/models/all"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":null,"family":"gpt-5.5","id":"gpt-5.5-pro","labId":"openai","license":"proprietary","name":"GPT-5.5 Pro","releasedOn":null,"retirementOn":null,"status":"active","url":"https://developers.openai.com/api/docs/models/all"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":1050000,"family":"GPT-5.6","id":"gpt-5.6-luna","labId":"openai","license":"proprietary","name":"GPT-5.6 Luna","releasedOn":null,"retirementOn":null,"status":"active","url":"https://developers.openai.com/api/docs/models/all"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":1050000,"family":"GPT-5.6","id":"gpt-5.6-sol","labId":"openai","license":"proprietary","name":"GPT-5.6 Sol","releasedOn":null,"retirementOn":null,"status":"active","url":"https://developers.openai.com/api/docs/models/all"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":1050000,"family":"GPT-5.6","id":"gpt-5.6-terra","labId":"openai","license":"proprietary","name":"GPT-5.6 Terra","releasedOn":null,"retirementOn":null,"status":"active","url":"https://developers.openai.com/api/docs/models/all"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":1050000,"family":"GPT-6","id":"gpt-6-astra","labId":"openai","license":"proprietary","name":"GPT-6 Astra","releasedOn":null,"retirementOn":null,"status":"active","url":"https://developers.openai.com/api/docs/models/all"},{"auditedOn":"2026-09-22","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":1050000,"family":"GPT-6","id":"gpt-6-luna","inputUsdPerMillion":0.1,"labId":"openai","license":"proprietary","name":"GPT-6 Luna","outputUsdPerMillion":0.5,"priceNote":"Standard rate for requests up to 272,000 input tokens. Longer requests are priced at 2× input and 1.5× output for the whole request.","releasedOn":"2026-09-22","retirementOn":null,"status":"active","url":"https://developers.openai.com/api/docs/models/gpt-6-luna"},{"auditedOn":"2026-09-29","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":1050000,"family":"GPT-6.1","id":"gpt-6.1-sol","inputUsdPerMillion":2,"labId":"openai","license":"proprietary","name":"GPT-6.1 Sol","outputUsdPerMillion":10,"priceNote":"Standard rate for requests up to 272,000 input tokens. Longer requests are priced at 2× input and 1.5× output for the whole request.","releasedOn":"2026-09-29","retirementOn":null,"status":"active","url":"https://developers.openai.com/api/docs/models/gpt-6.1-sol"},{"auditedOn":"2026-09-22","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":1050000,"family":"GPT-6","id":"gpt-6-sol","inputUsdPerMillion":2,"labId":"openai","license":"proprietary","name":"GPT-6 Sol","outputUsdPerMillion":10,"priceNote":"Standard rate for requests up to 272,000 input tokens. Longer requests are priced at 2× input and 1.5× output for the whole request.","releasedOn":"2026-09-22","retirementOn":null,"status":"active","url":"https://developers.openai.com/api/docs/models/gpt-6-sol"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"gpt-oss","id":"gpt-oss-120b","labId":"openai","license":"Apache 2.0","name":"gpt-oss-120b","releasedOn":null,"retirementOn":null,"status":"active","url":"https://developers.openai.com/api/docs/models/all"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"gpt-oss","id":"gpt-oss-20b","labId":"openai","license":"Apache 2.0","name":"gpt-oss-20b","releasedOn":null,"retirementOn":null,"status":"active","url":"https://developers.openai.com/api/docs/models/all"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":null,"family":"o1","id":"o1","labId":"openai","license":"proprietary","name":"o1","releasedOn":null,"retirementOn":"2026-10-23","status":"deprecated","url":"https://developers.openai.com/api/docs/deprecations"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":null,"family":"o1-pro","id":"o1-pro","labId":"openai","license":"proprietary","name":"o1-pro","releasedOn":null,"retirementOn":"2026-10-23","status":"deprecated","url":"https://developers.openai.com/api/docs/deprecations"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":null,"family":"o3","id":"o3","labId":"openai","license":"proprietary","name":"o3","releasedOn":null,"retirementOn":"2026-12-11","status":"deprecated","url":"https://developers.openai.com/api/docs/models/all"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":null,"family":"o3-mini","id":"o3-mini","labId":"openai","license":"proprietary","name":"o3-mini","releasedOn":null,"retirementOn":"2026-10-23","status":"deprecated","url":"https://developers.openai.com/api/docs/deprecations"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":null,"family":"o3-pro","id":"o3-pro","labId":"openai","license":"proprietary","name":"o3-pro","releasedOn":null,"retirementOn":"2026-12-11","status":"deprecated","url":"https://developers.openai.com/api/docs/models/all"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":null,"family":"o4-mini","id":"o4-mini","labId":"openai","license":"proprietary","name":"o4-mini","releasedOn":null,"retirementOn":"2026-10-23","status":"deprecated","url":"https://developers.openai.com/api/docs/deprecations"},{"auditedOn":"2026-09-06","availability":"Documented provider API; account and region restrictions may apply","contextTokens":null,"family":"qvq","id":"qvq-max","labId":"qwen","license":"proprietary","name":"QVQ-max","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://www.alibabacloud.com/help/en/model-studio/model-pricing"},{"auditedOn":"2026-09-06","availability":"Documented provider API; account and region restrictions may apply","contextTokens":null,"family":"qvq","id":"qvq-plus","labId":"qwen","license":"proprietary","name":"QVQ-plus","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://www.alibabacloud.com/help/en/model-studio/model-pricing"},{"auditedOn":"2026-09-06","availability":"Documented provider API; account and region restrictions may apply","contextTokens":null,"family":"qwen3.8","id":"qwen-3.8-max","labId":"qwen","license":"proprietary","name":"Qwen 3.8-Max","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://www.alibabacloud.com/help/en/model-studio/model-pricing"},{"auditedOn":"2026-09-06","availability":"Documented provider API; account and region restrictions may apply","contextTokens":null,"family":"qwen","id":"qwen-coder-plus","labId":"qwen","license":"proprietary","name":"Qwen-coder-plus","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://www.alibabacloud.com/help/en/model-studio/model-pricing"},{"auditedOn":"2026-09-06","availability":"Documented provider API; account and region restrictions may apply","contextTokens":null,"family":"qwen","id":"qwen-coder-turbo","labId":"qwen","license":"proprietary","name":"Qwen-coder-turbo","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://www.alibabacloud.com/help/en/model-studio/model-pricing"},{"auditedOn":"2026-09-06","availability":"Documented provider API; account and region restrictions may apply","contextTokens":null,"family":"qwen","id":"qwen-flash","labId":"qwen","license":"proprietary","name":"Qwen-flash","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://www.alibabacloud.com/help/en/model-studio/model-pricing"},{"auditedOn":"2026-09-06","availability":"Documented provider API; account and region restrictions may apply","contextTokens":null,"family":"qwen","id":"qwen-flash-character","labId":"qwen","license":"proprietary","name":"Qwen-flash-character","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://www.alibabacloud.com/help/en/model-studio/model-pricing"},{"auditedOn":"2026-09-06","availability":"Documented provider API; account and region restrictions may apply","contextTokens":null,"family":"qwen","id":"qwen-long","labId":"qwen","license":"proprietary","name":"Qwen-long","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://www.alibabacloud.com/help/en/model-studio/model-pricing"},{"auditedOn":"2026-09-06","availability":"Documented provider API; account and region restrictions may apply","contextTokens":null,"family":"qwen","id":"qwen-math-plus","labId":"qwen","license":"proprietary","name":"Qwen-math-plus","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://www.alibabacloud.com/help/en/model-studio/model-pricing"},{"auditedOn":"2026-09-06","availability":"Documented provider API; account and region restrictions may apply","contextTokens":null,"family":"qwen","id":"qwen-math-turbo","labId":"qwen","license":"proprietary","name":"Qwen-math-turbo","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://www.alibabacloud.com/help/en/model-studio/model-pricing"},{"auditedOn":"2026-09-06","availability":"Documented provider API; account and region restrictions may apply","contextTokens":null,"family":"qwen","id":"qwen-max","labId":"qwen","license":"proprietary","name":"Qwen-max","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://www.alibabacloud.com/help/en/model-studio/model-pricing"},{"auditedOn":"2026-09-06","availability":"Documented provider API; account and region restrictions may apply","contextTokens":null,"family":"qwen","id":"qwen-omni-turbo","labId":"qwen","license":"proprietary","name":"Qwen-omni-turbo","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://www.alibabacloud.com/help/en/model-studio/model-pricing"},{"auditedOn":"2026-09-06","availability":"Documented provider API; account and region restrictions may apply","contextTokens":null,"family":"qwen","id":"qwen-plus","labId":"qwen","license":"proprietary","name":"Qwen-plus","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://www.alibabacloud.com/help/en/model-studio/model-pricing"},{"auditedOn":"2026-09-06","availability":"Documented provider API; account and region restrictions may apply","contextTokens":null,"family":"qwen","id":"qwen-plus-character","labId":"qwen","license":"proprietary","name":"Qwen-plus-character","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://www.alibabacloud.com/help/en/model-studio/model-pricing"},{"auditedOn":"2026-09-06","availability":"Documented provider API; account and region restrictions may apply","contextTokens":null,"family":"qwen","id":"qwen-plus-character-ja","labId":"qwen","license":"proprietary","name":"Qwen-plus-character-ja","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://www.alibabacloud.com/help/en/model-studio/model-pricing"},{"auditedOn":"2026-09-06","availability":"Documented provider API; account and region restrictions may apply","contextTokens":null,"family":"qwen","id":"qwen-turbo","labId":"qwen","license":"proprietary","name":"Qwen-turbo","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://www.alibabacloud.com/help/en/model-studio/model-pricing"},{"auditedOn":"2026-09-06","availability":"Documented provider API; account and region restrictions may apply","contextTokens":null,"family":"qwen","id":"qwen-vl-max","labId":"qwen","license":"proprietary","name":"Qwen-vl-max","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://www.alibabacloud.com/help/en/model-studio/model-pricing"},{"auditedOn":"2026-09-06","availability":"Documented provider API; account and region restrictions may apply","contextTokens":null,"family":"qwen","id":"qwen-vl-plus","labId":"qwen","license":"proprietary","name":"Qwen-vl-plus","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://www.alibabacloud.com/help/en/model-studio/model-pricing"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Qwen3","id":"qwen3-0.6b","labId":"qwen","license":"apache-2.0","name":"Qwen3-0.6B","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/Qwen/Qwen3-0.6B"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Qwen3","id":"qwen3-1.7b","labId":"qwen","license":"apache-2.0","name":"Qwen3-1.7B","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/Qwen/Qwen3-1.7B"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Qwen3","id":"qwen3-14b","labId":"qwen","license":"apache-2.0","name":"Qwen3-14B","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/Qwen/Qwen3-14B"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Qwen3","id":"qwen3-235b-a22b","labId":"qwen","license":"apache-2.0","name":"Qwen3-235B-A22B","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/Qwen/Qwen3-235B-A22B"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Qwen3","id":"qwen3-235b-a22b-instruct-2507","labId":"qwen","license":"apache-2.0","name":"Qwen3-235B-A22B-Instruct-2507","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/Qwen/Qwen3-235B-A22B-Instruct-2507"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Qwen3","id":"qwen3-235b-a22b-thinking-2507","labId":"qwen","license":"apache-2.0","name":"Qwen3-235B-A22B-Thinking-2507","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/Qwen/Qwen3-235B-A22B-Thinking-2507"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Qwen3","id":"qwen3-30b-a3b","labId":"qwen","license":"apache-2.0","name":"Qwen3-30B-A3B","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/Qwen/Qwen3-30B-A3B"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Qwen3","id":"qwen3-30b-a3b-instruct-2507","labId":"qwen","license":"apache-2.0","name":"Qwen3-30B-A3B-Instruct-2507","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/Qwen/Qwen3-30B-A3B-Instruct-2507"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Qwen3","id":"qwen3-30b-a3b-thinking-2507","labId":"qwen","license":"apache-2.0","name":"Qwen3-30B-A3B-Thinking-2507","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/Qwen/Qwen3-30B-A3B-Thinking-2507"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Qwen3","id":"qwen3-32b","labId":"qwen","license":"apache-2.0","name":"Qwen3-32B","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/Qwen/Qwen3-32B"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Qwen3","id":"qwen3-4b","labId":"qwen","license":"apache-2.0","name":"Qwen3-4B","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/Qwen/Qwen3-4B"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Qwen3","id":"qwen3-4b-instruct-2507","labId":"qwen","license":"apache-2.0","name":"Qwen3-4B-Instruct-2507","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/Qwen/Qwen3-4B-Instruct-2507"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Qwen3","id":"qwen3-4b-thinking-2507","labId":"qwen","license":"apache-2.0","name":"Qwen3-4B-Thinking-2507","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/Qwen/Qwen3-4B-Thinking-2507"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Qwen3","id":"qwen3-8b","labId":"qwen","license":"apache-2.0","name":"Qwen3-8B","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/Qwen/Qwen3-8B"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Qwen3","id":"qwen3-coder-30b-a3b-instruct","labId":"qwen","license":"apache-2.0","name":"Qwen3-Coder-30B-A3B-Instruct","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/Qwen/Qwen3-Coder-30B-A3B-Instruct"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Qwen3","id":"qwen3-coder-480b-a35b-instruct","labId":"qwen","license":"apache-2.0","name":"Qwen3-Coder-480B-A35B-Instruct","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/Qwen/Qwen3-Coder-480B-A35B-Instruct"},{"auditedOn":"2026-09-06","availability":"Documented provider API; account and region restrictions may apply","contextTokens":null,"family":"qwen3","id":"qwen3-coder-flash","labId":"qwen","license":"proprietary","name":"Qwen3-coder-flash","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://www.alibabacloud.com/help/en/model-studio/model-pricing"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Qwen3","id":"qwen3-coder-next","labId":"qwen","license":"apache-2.0","name":"Qwen3-Coder-Next","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/Qwen/Qwen3-Coder-Next"},{"auditedOn":"2026-09-06","availability":"Documented provider API; account and region restrictions may apply","contextTokens":null,"family":"qwen3","id":"qwen3-coder-plus","labId":"qwen","license":"proprietary","name":"Qwen3-coder-plus","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://www.alibabacloud.com/help/en/model-studio/model-pricing"},{"auditedOn":"2026-09-06","availability":"Documented provider API; account and region restrictions may apply","contextTokens":null,"family":"qwen3","id":"qwen3-max","labId":"qwen","license":"proprietary","name":"Qwen3-max","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://www.alibabacloud.com/help/en/model-studio/model-pricing"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Qwen3","id":"qwen3-next-80b-a3b-instruct","labId":"qwen","license":"apache-2.0","name":"Qwen3-Next-80B-A3B-Instruct","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/Qwen/Qwen3-Next-80B-A3B-Instruct"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Qwen3","id":"qwen3-next-80b-a3b-thinking","labId":"qwen","license":"apache-2.0","name":"Qwen3-Next-80B-A3B-Thinking","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/Qwen/Qwen3-Next-80B-A3B-Thinking"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Qwen3","id":"qwen3-omni-30b-a3b-instruct","labId":"qwen","license":null,"name":"Qwen3-Omni-30B-A3B-Instruct","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/Qwen/Qwen3-Omni-30B-A3B-Instruct"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Qwen3","id":"qwen3-omni-30b-a3b-thinking","labId":"qwen","license":null,"name":"Qwen3-Omni-30B-A3B-Thinking","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/Qwen/Qwen3-Omni-30B-A3B-Thinking"},{"auditedOn":"2026-09-06","availability":"Documented provider API; account and region restrictions may apply","contextTokens":null,"family":"qwen3","id":"qwen3-omni-flash","labId":"qwen","license":"proprietary","name":"Qwen3-omni-flash","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://www.alibabacloud.com/help/en/model-studio/model-pricing"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Qwen3","id":"qwen3-vl-235b-a22b-instruct","labId":"qwen","license":"apache-2.0","name":"Qwen3-VL-235B-A22B-Instruct","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/Qwen/Qwen3-VL-235B-A22B-Instruct"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Qwen3","id":"qwen3-vl-235b-a22b-thinking","labId":"qwen","license":"apache-2.0","name":"Qwen3-VL-235B-A22B-Thinking","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/Qwen/Qwen3-VL-235B-A22B-Thinking"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Qwen3","id":"qwen3-vl-2b-instruct","labId":"qwen","license":"apache-2.0","name":"Qwen3-VL-2B-Instruct","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/Qwen/Qwen3-VL-2B-Instruct"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Qwen3","id":"qwen3-vl-2b-thinking","labId":"qwen","license":"apache-2.0","name":"Qwen3-VL-2B-Thinking","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/Qwen/Qwen3-VL-2B-Thinking"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Qwen3","id":"qwen3-vl-30b-a3b-instruct","labId":"qwen","license":"apache-2.0","name":"Qwen3-VL-30B-A3B-Instruct","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/Qwen/Qwen3-VL-30B-A3B-Instruct"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Qwen3","id":"qwen3-vl-30b-a3b-thinking","labId":"qwen","license":"apache-2.0","name":"Qwen3-VL-30B-A3B-Thinking","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/Qwen/Qwen3-VL-30B-A3B-Thinking"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Qwen3","id":"qwen3-vl-32b-instruct","labId":"qwen","license":"apache-2.0","name":"Qwen3-VL-32B-Instruct","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/Qwen/Qwen3-VL-32B-Instruct"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Qwen3","id":"qwen3-vl-32b-thinking","labId":"qwen","license":"apache-2.0","name":"Qwen3-VL-32B-Thinking","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/Qwen/Qwen3-VL-32B-Thinking"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Qwen3","id":"qwen3-vl-4b-instruct","labId":"qwen","license":"apache-2.0","name":"Qwen3-VL-4B-Instruct","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/Qwen/Qwen3-VL-4B-Instruct"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Qwen3","id":"qwen3-vl-4b-thinking","labId":"qwen","license":"apache-2.0","name":"Qwen3-VL-4B-Thinking","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/Qwen/Qwen3-VL-4B-Thinking"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Qwen3","id":"qwen3-vl-8b-instruct","labId":"qwen","license":"apache-2.0","name":"Qwen3-VL-8B-Instruct","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/Qwen/Qwen3-VL-8B-Instruct"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Qwen3","id":"qwen3-vl-8b-thinking","labId":"qwen","license":"apache-2.0","name":"Qwen3-VL-8B-Thinking","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/Qwen/Qwen3-VL-8B-Thinking"},{"auditedOn":"2026-09-06","availability":"Documented provider API; account and region restrictions may apply","contextTokens":null,"family":"qwen3","id":"qwen3-vl-flash","labId":"qwen","license":"proprietary","name":"Qwen3-vl-flash","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://www.alibabacloud.com/help/en/model-studio/model-pricing"},{"auditedOn":"2026-09-06","availability":"Documented provider API; account and region restrictions may apply","contextTokens":null,"family":"qwen3","id":"qwen3-vl-plus","labId":"qwen","license":"proprietary","name":"Qwen3-vl-plus","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://www.alibabacloud.com/help/en/model-studio/model-pricing"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Qwen3.5","id":"qwen3.5-0.8b","labId":"qwen","license":"apache-2.0","name":"Qwen3.5-0.8B","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/Qwen/Qwen3.5-0.8B"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Qwen3.5","id":"qwen3.5-122b-a10b","labId":"qwen","license":"apache-2.0","name":"Qwen3.5-122B-A10B","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/Qwen/Qwen3.5-122B-A10B"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Qwen3.5","id":"qwen3.5-27b","labId":"qwen","license":"apache-2.0","name":"Qwen3.5-27B","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/Qwen/Qwen3.5-27B"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Qwen3.5","id":"qwen3.5-2b","labId":"qwen","license":"apache-2.0","name":"Qwen3.5-2B","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/Qwen/Qwen3.5-2B"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Qwen3.5","id":"qwen3.5-35b-a3b","labId":"qwen","license":"apache-2.0","name":"Qwen3.5-35B-A3B","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/Qwen/Qwen3.5-35B-A3B"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Qwen3.5","id":"qwen3.5-397b-a17b","labId":"qwen","license":"apache-2.0","name":"Qwen3.5-397B-A17B","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/Qwen/Qwen3.5-397B-A17B"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Qwen3.5","id":"qwen3.5-4b","labId":"qwen","license":"apache-2.0","name":"Qwen3.5-4B","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/Qwen/Qwen3.5-4B"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Qwen3.5","id":"qwen3.5-9b","labId":"qwen","license":"apache-2.0","name":"Qwen3.5-9B","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/Qwen/Qwen3.5-9B"},{"auditedOn":"2026-09-06","availability":"Documented provider API; account and region restrictions may apply","contextTokens":null,"family":"qwen3.5","id":"qwen3.5-flash","labId":"qwen","license":"proprietary","name":"Qwen3.5-flash","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://www.alibabacloud.com/help/en/model-studio/model-pricing"},{"auditedOn":"2026-09-06","availability":"Documented provider API; account and region restrictions may apply","contextTokens":null,"family":"qwen3.5","id":"qwen3.5-omni-flash","labId":"qwen","license":"proprietary","name":"Qwen3.5-omni-flash","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://www.alibabacloud.com/help/en/model-studio/model-pricing"},{"auditedOn":"2026-09-06","availability":"Documented provider API; account and region restrictions may apply","contextTokens":null,"family":"qwen3.5","id":"qwen3.5-omni-plus","labId":"qwen","license":"proprietary","name":"Qwen3.5-omni-plus","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://www.alibabacloud.com/help/en/model-studio/model-pricing"},{"auditedOn":"2026-09-06","availability":"Documented provider API; account and region restrictions may apply","contextTokens":null,"family":"qwen3.5","id":"qwen3.5-plus","labId":"qwen","license":"proprietary","name":"Qwen3.5-plus","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://www.alibabacloud.com/help/en/model-studio/model-pricing"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Qwen3.6","id":"qwen3.6-27b","labId":"qwen","license":"apache-2.0","name":"Qwen3.6-27B","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/Qwen/Qwen3.6-27B"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Qwen3.6","id":"qwen3.6-35b-a3b","labId":"qwen","license":"apache-2.0","name":"Qwen3.6-35B-A3B","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/Qwen/Qwen3.6-35B-A3B"},{"auditedOn":"2026-09-06","availability":"Documented provider API; account and region restrictions may apply","contextTokens":null,"family":"qwen3.6","id":"qwen3.6-flash","labId":"qwen","license":"proprietary","name":"Qwen3.6-flash","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://www.alibabacloud.com/help/en/model-studio/model-pricing"},{"auditedOn":"2026-09-06","availability":"Documented provider API; account and region restrictions may apply","contextTokens":null,"family":"qwen3.6","id":"qwen3.6-plus","labId":"qwen","license":"proprietary","name":"Qwen3.6-plus","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://www.alibabacloud.com/help/en/model-studio/model-pricing"},{"auditedOn":"2026-09-06","availability":"Documented provider API; account and region restrictions may apply","contextTokens":null,"family":"qwen3.7","id":"qwen3.7-flash","labId":"qwen","license":"proprietary","name":"Qwen3.7-flash","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://www.alibabacloud.com/help/en/model-studio/model-pricing"},{"auditedOn":"2026-09-06","availability":"Documented provider API; account and region restrictions may apply","contextTokens":null,"family":"qwen3.7","id":"qwen3.7-max","labId":"qwen","license":"proprietary","name":"Qwen3.7-max","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://www.alibabacloud.com/help/en/model-studio/model-pricing"},{"auditedOn":"2026-09-06","availability":"Documented provider API; account and region restrictions may apply","contextTokens":null,"family":"qwen3.7","id":"qwen3.7-plus","labId":"qwen","license":"proprietary","name":"Qwen3.7-plus","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://www.alibabacloud.com/help/en/model-studio/model-pricing"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Qwen3.8","id":"qwen3.8-2.4t-a95b","labId":"qwen","license":null,"name":"Qwen3.8-2.4T-A95B","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Qwen3.8","id":"qwen3.8-27b","labId":"qwen","license":"apache-2.0","name":"Qwen3.8-27B","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/Qwen/Qwen3.8-27B"},{"auditedOn":"2026-09-06","availability":"Documented provider API; account and region restrictions may apply","contextTokens":null,"family":"qwen3.8","id":"qwen3.8-flash","labId":"qwen","license":"proprietary","name":"Qwen3.8-flash","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://www.alibabacloud.com/help/en/model-studio/model-pricing"},{"auditedOn":"2026-09-06","availability":"Open weights available for self-hosting","contextTokens":null,"family":"Qwen3.8","id":"qwen3.8-flash-next","labId":"qwen","license":null,"name":"Qwen3.8-Flash-Next","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/Qwen/Qwen3.8-Flash-Next"},{"auditedOn":"2026-09-06","availability":"Documented provider API; account and region restrictions may apply","contextTokens":null,"family":"qwq","id":"qwq-plus","labId":"qwen","license":"proprietary","name":"QwQ-plus","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://www.alibabacloud.com/help/en/model-studio/model-pricing"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":1000000,"family":"Grok 4","id":"grok-4.20-0309-non-reasoning","labId":"xai","license":"proprietary","name":"Grok-4.20-0309-non-reasoning","releasedOn":null,"retirementOn":null,"status":"active","url":"https://docs.x.ai/developers/pricing"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":1000000,"family":"Grok 4","id":"grok-4.20-0309-reasoning","labId":"xai","license":"proprietary","name":"Grok-4.20-0309-reasoning","releasedOn":null,"retirementOn":null,"status":"active","url":"https://docs.x.ai/developers/pricing"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":1000000,"family":"Grok 4","id":"grok-4.20-multi-agent-0309","labId":"xai","license":"proprietary","name":"Grok-4.20-multi-agent-0309","releasedOn":null,"retirementOn":null,"status":"active","url":"https://docs.x.ai/developers/pricing"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":1000000,"family":"Grok 4","id":"grok-4.3","labId":"xai","license":"proprietary","name":"Grok-4.3","releasedOn":null,"retirementOn":null,"status":"active","url":"https://docs.x.ai/developers/pricing"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":500000,"family":"Grok 4","id":"grok-4.5","labId":"xai","license":"proprietary","name":"Grok-4.5","releasedOn":null,"retirementOn":null,"status":"active","url":"https://docs.x.ai/developers/pricing"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":500000,"family":"Grok 4","id":"grok-4.6","labId":"xai","license":"proprietary","name":"Grok 4.6","releasedOn":null,"retirementOn":null,"status":"active","url":"https://docs.x.ai/developers/pricing"},{"auditedOn":"2026-09-21","availability":"Available through the Grok API, Grok Build, Cursor, and supported third-party platforms; account and region restrictions may apply","contextTokens":500000,"family":"Grok 4","id":"grok-4.7","labId":"xai","license":"proprietary","name":"Grok 4.7","releasedOn":"2026-09-21","retirementOn":null,"status":"active","url":"https://docs.x.ai/developers/models/grok-4.7"},{"auditedOn":"2026-09-06","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":256000,"family":"Grok 4","id":"grok-build-0.1","labId":"xai","license":"proprietary","name":"Grok-build-0.1","releasedOn":null,"retirementOn":null,"status":"active","url":"https://docs.x.ai/developers/pricing"},{"auditedOn":"2026-09-06","availability":"Documented provider API; downloadable official weights","contextTokens":1000000,"family":"GLM","id":"glm-5.3-flash","labId":"zai","license":"MIT","name":"GLM-5.3-Flash","releasedOn":null,"retirementOn":null,"status":"active","url":"https://docs.z.ai/guides/overview/overview"},{"auditedOn":"2026-09-06","availability":"Documented provider API; downloadable official weights","contextTokens":1000000,"family":"GLM","id":"glm-5.3","labId":"zai","license":null,"name":"GLM-5.3","releasedOn":null,"retirementOn":null,"status":"active","url":"https://docs.z.ai/guides/overview/overview"},{"auditedOn":"2026-09-06","availability":"Documented provider API; downloadable official weights","contextTokens":1000000,"family":"GLM","id":"glm-5.2","labId":"zai","license":"MIT","name":"GLM-5.2","releasedOn":null,"retirementOn":null,"status":"active","url":"https://docs.z.ai/guides/overview/overview"},{"auditedOn":"2026-09-06","availability":"Documented provider API; downloadable official weights","contextTokens":200000,"family":"GLM","id":"glm-5.1","labId":"zai","license":"MIT","name":"GLM-5.1","releasedOn":null,"retirementOn":null,"status":"active","url":"https://docs.z.ai/guides/overview/overview"},{"auditedOn":"2026-09-06","availability":"Documented provider API; downloadable official weights","contextTokens":200000,"family":"GLM","id":"glm-5","labId":"zai","license":"MIT","name":"GLM-5","releasedOn":null,"retirementOn":null,"status":"active","url":"https://docs.z.ai/guides/overview/overview"},{"auditedOn":"2026-09-06","availability":"Documented provider API; downloadable official weights","contextTokens":200000,"family":"GLM","id":"glm-4.7","labId":"zai","license":"MIT","name":"GLM-4.7","releasedOn":null,"retirementOn":null,"status":"active","url":"https://docs.z.ai/guides/overview/overview"},{"auditedOn":"2026-09-06","availability":"Documented provider API","contextTokens":200000,"family":"GLM","id":"glm-4.7-flashx","labId":"zai","license":null,"name":"GLM-4.7-FlashX","releasedOn":null,"retirementOn":null,"status":"active","url":"https://docs.z.ai/guides/overview/overview"},{"auditedOn":"2026-09-06","availability":"Documented provider API; downloadable official weights","contextTokens":200000,"family":"GLM","id":"glm-4.6","labId":"zai","license":"MIT","name":"GLM-4.6","releasedOn":null,"retirementOn":null,"status":"active","url":"https://docs.z.ai/guides/overview/overview"},{"auditedOn":"2026-09-06","availability":"Documented provider API; downloadable official weights","contextTokens":128000,"family":"GLM","id":"glm-4.5","labId":"zai","license":"MIT","name":"GLM-4.5","releasedOn":null,"retirementOn":null,"status":"active","url":"https://docs.z.ai/guides/overview/overview"},{"auditedOn":"2026-09-06","availability":"Documented provider API","contextTokens":128000,"family":"GLM","id":"glm-4.5-x","labId":"zai","license":null,"name":"GLM-4.5-X","releasedOn":null,"retirementOn":null,"status":"active","url":"https://docs.z.ai/guides/overview/overview"},{"auditedOn":"2026-09-06","availability":"Documented provider API; downloadable official weights","contextTokens":128000,"family":"GLM","id":"glm-4.5-air","labId":"zai","license":"MIT","name":"GLM-4.5-Air","releasedOn":null,"retirementOn":null,"status":"active","url":"https://docs.z.ai/guides/overview/overview"},{"auditedOn":"2026-09-06","availability":"Documented provider API","contextTokens":128000,"family":"GLM","id":"glm-4.5-airx","labId":"zai","license":null,"name":"GLM-4.5-AirX","releasedOn":null,"retirementOn":null,"status":"active","url":"https://docs.z.ai/guides/overview/overview"},{"auditedOn":"2026-09-06","availability":"Documented provider API","contextTokens":128000,"family":"GLM","id":"glm-4-32b-0414-128k","labId":"zai","license":null,"name":"GLM-4-32B-0414-128K","releasedOn":null,"retirementOn":null,"status":"active","url":"https://docs.z.ai/guides/overview/overview"},{"auditedOn":"2026-09-06","availability":"Documented provider API; downloadable official weights","contextTokens":200000,"family":"GLM","id":"glm-4.7-flash","labId":"zai","license":"MIT","name":"GLM-4.7-Flash","releasedOn":null,"retirementOn":null,"status":"active","url":"https://docs.z.ai/guides/overview/overview"},{"auditedOn":"2026-09-06","availability":"Documented provider API","contextTokens":200000,"family":"GLM","id":"glm-4.5-flash","labId":"zai","license":null,"name":"GLM-4.5-Flash","releasedOn":null,"retirementOn":null,"status":"active","url":"https://docs.z.ai/guides/overview/overview"},{"auditedOn":"2026-09-06","availability":"Documented provider API; downloadable official weights","contextTokens":128000,"family":"GLM","id":"glm-4.6v","labId":"zai","license":"MIT","name":"GLM-4.6V","releasedOn":null,"retirementOn":null,"status":"active","url":"https://docs.z.ai/guides/overview/overview"},{"auditedOn":"2026-09-06","availability":"Documented provider API","contextTokens":128000,"family":"GLM","id":"glm-4.6v-flashx","labId":"zai","license":null,"name":"GLM-4.6V-FlashX","releasedOn":null,"retirementOn":null,"status":"active","url":"https://docs.z.ai/guides/overview/overview"},{"auditedOn":"2026-09-06","availability":"Documented provider API; downloadable official weights","contextTokens":64000,"family":"GLM","id":"glm-4.5v","labId":"zai","license":"MIT","name":"GLM-4.5V","releasedOn":null,"retirementOn":null,"status":"active","url":"https://docs.z.ai/guides/overview/overview"},{"auditedOn":"2026-09-06","availability":"Documented provider API; downloadable official weights","contextTokens":128000,"family":"GLM","id":"glm-4.6v-flash","labId":"zai","license":"MIT","name":"GLM-4.6V-Flash","releasedOn":null,"retirementOn":null,"status":"active","url":"https://docs.z.ai/guides/overview/overview"},{"auditedOn":"2026-09-06","availability":"Documented provider API","contextTokens":200000,"family":"GLM","id":"glm-5-turbo","labId":"zai","license":null,"name":"GLM-5-Turbo","releasedOn":null,"retirementOn":null,"status":"active","url":"https://docs.z.ai/guides/llm/glm-5-turbo"},{"auditedOn":"2026-09-06","availability":"Documented provider API","contextTokens":200000,"family":"GLM","id":"glm-5v-turbo","labId":"zai","license":null,"name":"GLM-5V-Turbo","releasedOn":null,"retirementOn":null,"status":"active","url":"https://docs.z.ai/guides/vlm/glm-5v-turbo"},{"auditedOn":"2026-09-06","availability":"Downloadable official model weights","contextTokens":null,"family":"GLM","id":"glm-4.1v-9b-thinking","labId":"zai","license":"MIT","name":"GLM-4.1V-9B-Thinking","releasedOn":null,"retirementOn":null,"status":"active","url":"https://huggingface.co/zai-org/GLM-4.1V-9B-Thinking"},{"auditedOn":"2026-09-06","availability":"Downloadable official model weights","contextTokens":null,"family":"GLM","id":"glm-z1-rumination-32b-0414","labId":"zai","license":"MIT","name":"GLM-Z1-Rumination-32B-0414","releasedOn":null,"retirementOn":null,"status":"active","url":"https://huggingface.co/zai-org/GLM-Z1-Rumination-32B-0414"},{"auditedOn":"2026-09-06","availability":"Downloadable official model weights","contextTokens":null,"family":"GLM","id":"glm-z1-9b-0414","labId":"zai","license":"MIT","name":"GLM-Z1-9B-0414","releasedOn":null,"retirementOn":null,"status":"active","url":"https://huggingface.co/zai-org/GLM-Z1-9B-0414"},{"auditedOn":"2026-09-06","availability":"Downloadable official model weights","contextTokens":null,"family":"GLM","id":"glm-z1-32b-0414","labId":"zai","license":"MIT","name":"GLM-Z1-32B-0414","releasedOn":null,"retirementOn":null,"status":"active","url":"https://huggingface.co/zai-org/GLM-Z1-32B-0414"},{"auditedOn":"2026-09-06","availability":"Downloadable official model weights","contextTokens":null,"family":"GLM","id":"glm-4-9b-0414","labId":"zai","license":"MIT","name":"GLM-4-9B-0414","releasedOn":null,"retirementOn":null,"status":"active","url":"https://huggingface.co/zai-org/GLM-4-9B-0414"},{"auditedOn":"2026-09-06","availability":"Downloadable official model weights","contextTokens":null,"family":"GLM","id":"glm-4-32b-0414","labId":"zai","license":"MIT","name":"GLM-4-32B-0414","releasedOn":null,"retirementOn":null,"status":"active","url":"https://huggingface.co/zai-org/GLM-4-32B-0414"},{"auditedOn":"2026-09-06","availability":"Downloadable official model weights","contextTokens":null,"family":"GLM","id":"glm-4-9b-chat","labId":"zai","license":null,"name":"glm-4-9b-chat","releasedOn":null,"retirementOn":null,"status":"active","url":"https://huggingface.co/zai-org/glm-4-9b-chat"},{"auditedOn":"2026-09-06","availability":"Downloadable official model weights","contextTokens":null,"family":"GLM","id":"glm-4-9b-chat-1m","labId":"zai","license":null,"name":"glm-4-9b-chat-1m","releasedOn":null,"retirementOn":null,"status":"active","url":"https://huggingface.co/zai-org/glm-4-9b-chat-1m"},{"auditedOn":"2026-09-06","availability":"Downloadable official model weights","contextTokens":null,"family":"GLM","id":"glm-4v-9b","labId":"zai","license":null,"name":"glm-4v-9b","releasedOn":null,"retirementOn":null,"status":"active","url":"https://huggingface.co/zai-org/glm-4v-9b"},{"auditedOn":"2026-09-06","availability":"Downloadable official model weights","contextTokens":null,"family":"GLM","id":"glm-edge-1.5b-chat","labId":"zai","license":null,"name":"glm-edge-1.5b-chat","releasedOn":null,"retirementOn":null,"status":"active","url":"https://huggingface.co/zai-org/glm-edge-1.5b-chat"},{"auditedOn":"2026-09-06","availability":"Downloadable official model weights","contextTokens":null,"family":"GLM","id":"glm-edge-4b-chat","labId":"zai","license":null,"name":"glm-edge-4b-chat","releasedOn":null,"retirementOn":null,"status":"active","url":"https://huggingface.co/zai-org/glm-edge-4b-chat"},{"auditedOn":"2026-09-06","availability":"Downloadable official model weights","contextTokens":null,"family":"GLM","id":"glm-edge-v-2b","labId":"zai","license":null,"name":"glm-edge-v-2b","releasedOn":null,"retirementOn":null,"status":"active","url":"https://huggingface.co/zai-org/glm-edge-v-2b"},{"auditedOn":"2026-09-06","availability":"Downloadable official model weights","contextTokens":null,"family":"GLM","id":"glm-edge-v-5b","labId":"zai","license":null,"name":"glm-edge-v-5b","releasedOn":null,"retirementOn":null,"status":"active","url":"https://huggingface.co/zai-org/glm-edge-v-5b"},{"auditedOn":"2026-09-06","availability":"Downloadable official model weights","contextTokens":null,"family":"GLM","id":"chatglm-6b","labId":"zai","license":null,"name":"chatglm-6b","releasedOn":null,"retirementOn":null,"status":"active","url":"https://huggingface.co/zai-org/chatglm-6b"},{"auditedOn":"2026-09-06","availability":"Downloadable official model weights","contextTokens":null,"family":"GLM","id":"chatglm2-6b","labId":"zai","license":null,"name":"chatglm2-6b","releasedOn":null,"retirementOn":null,"status":"active","url":"https://huggingface.co/zai-org/chatglm2-6b"},{"auditedOn":"2026-09-06","availability":"Downloadable official model weights","contextTokens":null,"family":"GLM","id":"chatglm2-6b-32k","labId":"zai","license":null,"name":"chatglm2-6b-32k","releasedOn":null,"retirementOn":null,"status":"active","url":"https://huggingface.co/zai-org/chatglm2-6b-32k"},{"auditedOn":"2026-09-06","availability":"Downloadable official model weights","contextTokens":null,"family":"GLM","id":"chatglm3-6b","labId":"zai","license":null,"name":"chatglm3-6b","releasedOn":null,"retirementOn":null,"status":"active","url":"https://huggingface.co/zai-org/chatglm3-6b"},{"auditedOn":"2026-09-06","availability":"Downloadable official model weights","contextTokens":null,"family":"GLM","id":"chatglm3-6b-32k","labId":"zai","license":null,"name":"chatglm3-6b-32k","releasedOn":null,"retirementOn":null,"status":"active","url":"https://huggingface.co/zai-org/chatglm3-6b-32k"},{"auditedOn":"2026-09-06","availability":"Downloadable official model weights","contextTokens":null,"family":"GLM","id":"chatglm3-6b-128k","labId":"zai","license":null,"name":"chatglm3-6b-128k","releasedOn":null,"retirementOn":null,"status":"active","url":"https://huggingface.co/zai-org/chatglm3-6b-128k"},{"auditedOn":"2026-09-06","availability":"Downloadable official model weights","contextTokens":null,"family":"GLM","id":"codegeex4-all-9b","labId":"zai","license":null,"name":"codegeex4-all-9b","releasedOn":null,"retirementOn":null,"status":"active","url":"https://huggingface.co/zai-org/codegeex4-all-9b"},{"auditedOn":"2026-09-06","availability":"Downloadable official model weights","contextTokens":null,"family":"GLM","id":"autoglm-phone-9b","labId":"zai","license":"MIT","name":"AutoGLM-Phone-9B","releasedOn":null,"retirementOn":null,"status":"active","url":"https://huggingface.co/zai-org/AutoGLM-Phone-9B"},{"auditedOn":"2026-09-06","availability":"Downloadable official model weights","contextTokens":null,"family":"GLM","id":"autoglm-phone-9b-multilingual","labId":"zai","license":"MIT","name":"AutoGLM-Phone-9B-Multilingual","releasedOn":null,"retirementOn":null,"status":"active","url":"https://huggingface.co/zai-org/AutoGLM-Phone-9B-Multilingual"},{"auditedOn":"2026-09-06","availability":"Downloadable official model weights","contextTokens":null,"family":"GLM","id":"cogagent-9b-20241220","labId":"zai","license":null,"name":"cogagent-9b-20241220","releasedOn":null,"retirementOn":null,"status":"active","url":"https://huggingface.co/zai-org/cogagent-9b-20241220"},{"auditedOn":"2026-09-06","availability":"Downloadable official model weights","contextTokens":null,"family":"GLM","id":"codegeex2-6b","labId":"zai","license":null,"name":"codegeex2-6b","releasedOn":null,"retirementOn":null,"status":"active","url":"https://huggingface.co/zai-org/codegeex2-6b"},{"auditedOn":"2026-09-06","availability":"Downloadable official model weights","contextTokens":null,"family":"GLM","id":"visualglm-6b","labId":"zai","license":null,"name":"visualglm-6b","releasedOn":null,"retirementOn":null,"status":"active","url":"https://huggingface.co/zai-org/visualglm-6b"},{"auditedOn":"2026-09-06","availability":"Downloadable official model weights","contextTokens":null,"family":"GLM","id":"cogvlm-chat-hf","labId":"zai","license":"Apache-2.0","name":"cogvlm-chat-hf","releasedOn":null,"retirementOn":null,"status":"active","url":"https://huggingface.co/zai-org/cogvlm-chat-hf"},{"auditedOn":"2026-09-06","availability":"Downloadable official model weights","contextTokens":null,"family":"GLM","id":"cogagent-chat-hf","labId":"zai","license":"Apache-2.0","name":"cogagent-chat-hf","releasedOn":null,"retirementOn":null,"status":"active","url":"https://huggingface.co/zai-org/cogagent-chat-hf"},{"auditedOn":"2026-09-06","availability":"Downloadable official model weights","contextTokens":null,"family":"GLM","id":"cogagent-vqa-hf","labId":"zai","license":"Apache-2.0","name":"cogagent-vqa-hf","releasedOn":null,"retirementOn":null,"status":"active","url":"https://huggingface.co/zai-org/cogagent-vqa-hf"},{"auditedOn":"2026-09-06","availability":"Downloadable official model weights","contextTokens":null,"family":"GLM","id":"cogvlm2-llama3-chat-19b","labId":"zai","license":null,"name":"cogvlm2-llama3-chat-19B","releasedOn":null,"retirementOn":null,"status":"active","url":"https://huggingface.co/zai-org/cogvlm2-llama3-chat-19B"},{"auditedOn":"2026-09-06","availability":"Downloadable official model weights","contextTokens":null,"family":"GLM","id":"cogvlm2-llama3-chinese-chat-19b","labId":"zai","license":null,"name":"cogvlm2-llama3-chinese-chat-19B","releasedOn":null,"retirementOn":null,"status":"active","url":"https://huggingface.co/zai-org/cogvlm2-llama3-chinese-chat-19B"},{"auditedOn":"2026-09-06","availability":"Downloadable official model weights","contextTokens":null,"family":"GLM","id":"cogvlm2-video-llama3-chat","labId":"zai","license":null,"name":"cogvlm2-video-llama3-chat","releasedOn":null,"retirementOn":null,"status":"active","url":"https://huggingface.co/zai-org/cogvlm2-video-llama3-chat"},{"auditedOn":"2026-09-06","availability":"Documented provider API; downloadable official weights","contextTokens":1000000,"family":"MiniMax","id":"minimax-m3","labId":"minimax","license":null,"name":"MiniMax-M3","releasedOn":null,"retirementOn":null,"status":"active","url":"https://platform.minimax.io/docs/api-reference/text-anthropic-api"},{"auditedOn":"2026-09-06","availability":"Documented provider API; downloadable official weights","contextTokens":204800,"family":"MiniMax","id":"minimax-m2.7","labId":"minimax","license":null,"name":"MiniMax-M2.7","releasedOn":null,"retirementOn":null,"status":"active","url":"https://platform.minimax.io/docs/api-reference/text-anthropic-api"},{"auditedOn":"2026-09-06","availability":"Documented provider API; downloadable official weights","contextTokens":204800,"family":"MiniMax","id":"minimax-m2.5","labId":"minimax","license":null,"name":"MiniMax-M2.5","releasedOn":null,"retirementOn":null,"status":"legacy","url":"https://platform.minimax.io/docs/api-reference/text-anthropic-api"},{"auditedOn":"2026-09-06","availability":"Documented provider API; downloadable official weights","contextTokens":204800,"family":"MiniMax","id":"minimax-m2.1","labId":"minimax","license":null,"name":"MiniMax-M2.1","releasedOn":null,"retirementOn":null,"status":"legacy","url":"https://platform.minimax.io/docs/api-reference/text-anthropic-api"},{"auditedOn":"2026-09-06","availability":"Documented provider API; downloadable official weights","contextTokens":204800,"family":"MiniMax","id":"minimax-m2","labId":"minimax","license":null,"name":"MiniMax-M2","releasedOn":null,"retirementOn":null,"status":"legacy","url":"https://platform.minimax.io/docs/api-reference/text-anthropic-api"},{"auditedOn":"2026-09-06","availability":"Downloadable official model weights","contextTokens":null,"family":"MiniMax","id":"minimax-text-01","labId":"minimax","license":null,"name":"MiniMax-Text-01","releasedOn":null,"retirementOn":null,"status":"active","url":"https://huggingface.co/MiniMaxAI/MiniMax-Text-01"},{"auditedOn":"2026-09-06","availability":"Downloadable official model weights","contextTokens":null,"family":"MiniMax","id":"minimax-vl-01","labId":"minimax","license":null,"name":"MiniMax-VL-01","releasedOn":null,"retirementOn":null,"status":"active","url":"https://huggingface.co/MiniMaxAI/MiniMax-VL-01"},{"auditedOn":"2026-09-06","availability":"Downloadable official model weights","contextTokens":null,"family":"MiniMax","id":"minimax-m1-40k","labId":"minimax","license":"Apache-2.0","name":"MiniMax-M1-40k","releasedOn":null,"retirementOn":null,"status":"active","url":"https://huggingface.co/MiniMaxAI/MiniMax-M1-40k"},{"auditedOn":"2026-09-06","availability":"Downloadable official model weights","contextTokens":null,"family":"MiniMax","id":"minimax-m1-80k","labId":"minimax","license":"Apache-2.0","name":"MiniMax-M1-80k","releasedOn":null,"retirementOn":null,"status":"active","url":"https://huggingface.co/MiniMaxAI/MiniMax-M1-80k"},{"auditedOn":"2026-09-06","availability":"Downloadable official model weights","contextTokens":null,"family":"MiniMax","id":"synlogic-32b","labId":"minimax","license":"MIT","name":"SynLogic-32B","releasedOn":null,"retirementOn":null,"status":"active","url":"https://huggingface.co/MiniMaxAI/SynLogic-32B"},{"auditedOn":"2026-09-06","availability":"Downloadable official model weights","contextTokens":null,"family":"MiniMax","id":"synlogic-mix-3-32b","labId":"minimax","license":"MIT","name":"SynLogic-Mix-3-32B","releasedOn":null,"retirementOn":null,"status":"active","url":"https://huggingface.co/MiniMaxAI/SynLogic-Mix-3-32B"},{"auditedOn":"2026-09-06","availability":"Downloadable official model weights","contextTokens":null,"family":"MiniMax","id":"synlogic-7b","labId":"minimax","license":"MIT","name":"SynLogic-7B","releasedOn":null,"retirementOn":null,"status":"active","url":"https://huggingface.co/MiniMaxAI/SynLogic-7B"},{"auditedOn":"2026-09-06","availability":"Documented provider API","contextTokens":64000,"family":"MiniMax","id":"minimax-m2-her","labId":"minimax","license":null,"name":"MiniMax-M2-her","releasedOn":null,"retirementOn":null,"status":"active","url":"https://platform.minimax.io/docs/guides/text-chat"},{"auditedOn":"2026-09-06","availability":"Documented provider API; downloadable official weights","contextTokens":128000,"family":"Command","id":"command-a-plus-05-2026","labId":"cohere","license":"Apache-2.0","name":"Command A Plus 05 2026","releasedOn":null,"retirementOn":null,"status":"active","url":"https://docs.cohere.com/docs/models"},{"auditedOn":"2026-09-06","availability":"Documented provider API; downloadable official weights","contextTokens":256000,"family":"Command","id":"command-a-03-2025","labId":"cohere","license":"CC-BY-NC-4.0","name":"Command A 03 2025","releasedOn":null,"retirementOn":null,"status":"active","url":"https://docs.cohere.com/docs/models"},{"auditedOn":"2026-09-06","availability":"Documented provider API; downloadable official weights","contextTokens":128000,"family":"Command","id":"command-r7b-12-2024","labId":"cohere","license":"CC-BY-NC-4.0","name":"Command R7B 12 2024","releasedOn":null,"retirementOn":null,"status":"active","url":"https://docs.cohere.com/docs/models"},{"auditedOn":"2026-09-06","availability":"Documented provider API; downloadable official weights","contextTokens":8000,"family":"Command","id":"command-a-translate-08-2025","labId":"cohere","license":"CC-BY-NC-4.0","name":"Command A Translate 08 2025","releasedOn":null,"retirementOn":null,"status":"active","url":"https://docs.cohere.com/docs/models"},{"auditedOn":"2026-09-06","availability":"Documented provider API; downloadable official weights","contextTokens":256000,"family":"Command","id":"command-a-reasoning-08-2025","labId":"cohere","license":"CC-BY-NC-4.0","name":"Command A Reasoning 08 2025","releasedOn":null,"retirementOn":null,"status":"active","url":"https://docs.cohere.com/docs/models"},{"auditedOn":"2026-09-06","availability":"Documented provider API; downloadable official weights","contextTokens":128000,"family":"Command","id":"command-a-vision-07-2025","labId":"cohere","license":"CC-BY-NC-4.0","name":"Command A Vision 07 2025","releasedOn":null,"retirementOn":null,"status":"active","url":"https://docs.cohere.com/docs/models"},{"auditedOn":"2026-09-06","availability":"Documented provider API; downloadable official weights","contextTokens":128000,"family":"Command","id":"command-r-08-2024","labId":"cohere","license":"CC-BY-NC-4.0","name":"Command R 08 2024","releasedOn":null,"retirementOn":null,"status":"active","url":"https://docs.cohere.com/docs/models"},{"auditedOn":"2026-09-06","availability":"Documented provider API; downloadable official weights","contextTokens":128000,"family":"Command","id":"command-r-plus-08-2024","labId":"cohere","license":"CC-BY-NC-4.0","name":"Command R Plus 08 2024","releasedOn":null,"retirementOn":null,"status":"active","url":"https://docs.cohere.com/docs/models"},{"auditedOn":"2026-09-06","availability":"Provider API for existing users until retirement; downloadable official weights","contextTokens":128000,"family":"Command","id":"command-r-03-2024","labId":"cohere","license":"CC-BY-NC-4.0","name":"Command R 03 2024","releasedOn":null,"retirementOn":null,"status":"deprecated","url":"https://docs.cohere.com/docs/models"},{"auditedOn":"2026-09-06","availability":"Provider API for existing users until retirement; downloadable official weights","contextTokens":128000,"family":"Command","id":"command-r-plus-04-2024","labId":"cohere","license":"CC-BY-NC-4.0","name":"Command R Plus 04 2024","releasedOn":null,"retirementOn":null,"status":"deprecated","url":"https://docs.cohere.com/docs/models"},{"auditedOn":"2026-09-06","availability":"Provider API for existing users until retirement","contextTokens":4000,"family":"Command","id":"command-light","labId":"cohere","license":null,"name":"Command Light","releasedOn":null,"retirementOn":null,"status":"deprecated","url":"https://docs.cohere.com/docs/models"},{"auditedOn":"2026-09-06","availability":"Provider API for existing users until retirement","contextTokens":4000,"family":"Command","id":"command","labId":"cohere","license":null,"name":"Command","releasedOn":null,"retirementOn":null,"status":"deprecated","url":"https://docs.cohere.com/docs/models"},{"auditedOn":"2026-09-06","availability":"Documented provider API; downloadable official weights","contextTokens":8000,"family":"Aya","id":"tiny-aya-global","labId":"cohere","license":"CC-BY-NC-4.0","name":"Tiny Aya Global","releasedOn":null,"retirementOn":null,"status":"active","url":"https://docs.cohere.com/docs/models"},{"auditedOn":"2026-09-06","availability":"Documented provider API; downloadable official weights","contextTokens":8000,"family":"Aya","id":"tiny-aya-earth","labId":"cohere","license":"CC-BY-NC-4.0","name":"Tiny Aya Earth","releasedOn":null,"retirementOn":null,"status":"active","url":"https://docs.cohere.com/docs/models"},{"auditedOn":"2026-09-06","availability":"Documented provider API; downloadable official weights","contextTokens":8000,"family":"Aya","id":"tiny-aya-fire","labId":"cohere","license":"CC-BY-NC-4.0","name":"Tiny Aya Fire","releasedOn":null,"retirementOn":null,"status":"active","url":"https://docs.cohere.com/docs/models"},{"auditedOn":"2026-09-06","availability":"Documented provider API; downloadable official weights","contextTokens":8000,"family":"Aya","id":"tiny-aya-water","labId":"cohere","license":"CC-BY-NC-4.0","name":"Tiny Aya Water","releasedOn":null,"retirementOn":null,"status":"active","url":"https://docs.cohere.com/docs/models"},{"auditedOn":"2026-09-06","availability":"Documented provider API; downloadable official weights","contextTokens":128000,"family":"Aya","id":"c4ai-aya-expanse-32b","labId":"cohere","license":"CC-BY-NC-4.0","name":"C4Ai Aya Expanse 32B","releasedOn":null,"retirementOn":null,"status":"active","url":"https://docs.cohere.com/docs/models"},{"auditedOn":"2026-09-06","availability":"Documented provider API; downloadable official weights","contextTokens":16000,"family":"Aya","id":"c4ai-aya-vision-32b","labId":"cohere","license":"CC-BY-NC-4.0","name":"C4Ai Aya Vision 32B","releasedOn":null,"retirementOn":null,"status":"active","url":"https://docs.cohere.com/docs/models"},{"auditedOn":"2026-09-06","availability":"Downloadable official model weights","contextTokens":null,"family":"Command","id":"c4ai-command-r7b-arabic-02-2025","labId":"cohere","license":"CC-BY-NC-4.0","name":"c4ai-command-r7b-arabic-02-2025","releasedOn":null,"retirementOn":null,"status":"active","url":"https://huggingface.co/CohereLabs/c4ai-command-r7b-arabic-02-2025"},{"auditedOn":"2026-09-06","availability":"Downloadable official weights; provider API retired April4 2026","contextTokens":null,"family":"Aya","id":"c4ai-aya-expanse-8b","labId":"cohere","license":"CC-BY-NC-4.0","name":"aya-expanse-8b","releasedOn":null,"retirementOn":null,"status":"active","url":"https://huggingface.co/CohereLabs/aya-expanse-8b"},{"auditedOn":"2026-09-06","availability":"Downloadable official weights; provider API retired April4 2026","contextTokens":null,"family":"Aya","id":"c4ai-aya-vision-8b","labId":"cohere","license":"CC-BY-NC-4.0","name":"aya-vision-8b","releasedOn":null,"retirementOn":null,"status":"active","url":"https://huggingface.co/CohereLabs/aya-vision-8b"},{"auditedOn":"2026-09-06","availability":"Downloadable official model weights","contextTokens":null,"family":"Aya","id":"aya-23-8b","labId":"cohere","license":"CC-BY-NC-4.0","name":"aya-23-8B","releasedOn":null,"retirementOn":null,"status":"active","url":"https://huggingface.co/CohereLabs/aya-23-8B"},{"auditedOn":"2026-09-06","availability":"Downloadable official model weights","contextTokens":null,"family":"Aya","id":"aya-23-35b","labId":"cohere","license":"CC-BY-NC-4.0","name":"aya-23-35B","releasedOn":null,"retirementOn":null,"status":"active","url":"https://huggingface.co/CohereLabs/aya-23-35B"},{"auditedOn":"2026-09-06","availability":"Downloadable official model weights","contextTokens":null,"family":"Aya","id":"aya-101","labId":"cohere","license":"Apache-2.0","name":"aya-101","releasedOn":null,"retirementOn":null,"status":"active","url":"https://huggingface.co/CohereLabs/aya-101"},{"auditedOn":"2026-09-06","availability":"Downloadable official model weights; provider API and Model Vault","contextTokens":256000,"family":"North","id":"north-mini-code-1-0","labId":"cohere","license":"Apache-2.0","name":"North-Mini-Code-1.0","releasedOn":null,"retirementOn":null,"status":"active","url":"https://huggingface.co/CohereLabs/North-Mini-Code-1.0"},{"auditedOn":"2026-09-06","availability":"Downloadable official model weights","contextTokens":8000,"family":"North","id":"north-micro-vision-instruct","labId":"cohere","license":"Apache-2.0","name":"North-Micro-Vision-Instruct","releasedOn":null,"retirementOn":null,"status":"active","url":"https://huggingface.co/CohereLabs/North-Micro-Vision-Instruct"},{"auditedOn":"2026-09-06","availability":"Downloadable Microsoft instruction/reasoning weights","contextTokens":16384,"family":"Phi","id":"phi-4","labId":"microsoft","license":"mit","name":"phi-4","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/microsoft/phi-4"},{"auditedOn":"2026-09-06","availability":"Downloadable Microsoft instruction/reasoning weights","contextTokens":131072,"family":"Phi","id":"phi-4-mini-instruct","labId":"microsoft","license":"mit","name":"Phi-4-mini-instruct","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/microsoft/Phi-4-mini-instruct"},{"auditedOn":"2026-09-06","availability":"Downloadable Microsoft instruction/reasoning weights","contextTokens":16384,"family":"Phi","id":"phi-4-reasoning-vision-15b","labId":"microsoft","license":"mit","name":"Phi-4-reasoning-vision-15B","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/microsoft/Phi-4-reasoning-vision-15B"},{"auditedOn":"2026-09-06","availability":"Downloadable Microsoft instruction/reasoning weights","contextTokens":4096,"family":"Phi","id":"phi-3-mini-4k-instruct","labId":"microsoft","license":"mit","name":"Phi-3-mini-4k-instruct","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/microsoft/Phi-3-mini-4k-instruct"},{"auditedOn":"2026-09-06","availability":"Downloadable Microsoft instruction/reasoning weights","contextTokens":131072,"family":"Phi","id":"phi-3-mini-128k-instruct","labId":"microsoft","license":"mit","name":"Phi-3-mini-128k-instruct","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/microsoft/Phi-3-mini-128k-instruct"},{"auditedOn":"2026-09-06","availability":"Downloadable Microsoft instruction/reasoning weights","contextTokens":131072,"family":"Phi","id":"phi-3.5-mini-instruct","labId":"microsoft","license":"mit","name":"Phi-3.5-mini-instruct","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/microsoft/Phi-3.5-mini-instruct"},{"auditedOn":"2026-09-06","availability":"Downloadable Microsoft instruction/reasoning weights","contextTokens":131072,"family":"Phi","id":"phi-4-multimodal-instruct","labId":"microsoft","license":"mit","name":"Phi-4-multimodal-instruct","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/microsoft/Phi-4-multimodal-instruct"},{"auditedOn":"2026-09-06","availability":"Downloadable Microsoft instruction/reasoning weights","contextTokens":32768,"family":"Phi","id":"phi-4-reasoning","labId":"microsoft","license":"mit","name":"Phi-4-reasoning","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/microsoft/Phi-4-reasoning"},{"auditedOn":"2026-09-06","availability":"Downloadable Microsoft instruction/reasoning weights","contextTokens":4096,"family":"Phi","id":"phi-mini-moe-instruct","labId":"microsoft","license":"mit","name":"Phi-mini-MoE-instruct","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/microsoft/Phi-mini-MoE-instruct"},{"auditedOn":"2026-09-06","availability":"Downloadable Microsoft instruction/reasoning weights","contextTokens":4096,"family":"Phi","id":"phi-tiny-moe-instruct","labId":"microsoft","license":"mit","name":"Phi-tiny-MoE-instruct","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/microsoft/Phi-tiny-MoE-instruct"},{"auditedOn":"2026-09-06","availability":"Downloadable Microsoft instruction/reasoning weights","contextTokens":4096,"family":"Phi","id":"phi-3-medium-4k-instruct","labId":"microsoft","license":"mit","name":"Phi-3-medium-4k-instruct","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/microsoft/Phi-3-medium-4k-instruct"},{"auditedOn":"2026-09-06","availability":"Downloadable Microsoft instruction/reasoning weights","contextTokens":131072,"family":"Phi","id":"phi-3-medium-128k-instruct","labId":"microsoft","license":"mit","name":"Phi-3-medium-128k-instruct","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/microsoft/Phi-3-medium-128k-instruct"},{"auditedOn":"2026-09-06","availability":"Downloadable Microsoft instruction/reasoning weights","contextTokens":8192,"family":"Phi","id":"phi-3-small-8k-instruct","labId":"microsoft","license":"mit","name":"Phi-3-small-8k-instruct","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/microsoft/Phi-3-small-8k-instruct"},{"auditedOn":"2026-09-06","availability":"Downloadable Microsoft instruction/reasoning weights","contextTokens":131072,"family":"Phi","id":"phi-3-small-128k-instruct","labId":"microsoft","license":"mit","name":"Phi-3-small-128k-instruct","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/microsoft/Phi-3-small-128k-instruct"},{"auditedOn":"2026-09-06","availability":"Downloadable Microsoft instruction/reasoning weights","contextTokens":131072,"family":"Phi","id":"phi-3-vision-128k-instruct","labId":"microsoft","license":"mit","name":"Phi-3-vision-128k-instruct","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/microsoft/Phi-3-vision-128k-instruct"},{"auditedOn":"2026-09-06","availability":"Downloadable Microsoft instruction/reasoning weights","contextTokens":131072,"family":"Phi","id":"phi-3.5-vision-instruct","labId":"microsoft","license":"mit","name":"Phi-3.5-vision-instruct","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/microsoft/Phi-3.5-vision-instruct"},{"auditedOn":"2026-09-06","availability":"Downloadable Microsoft instruction/reasoning weights","contextTokens":131072,"family":"Phi","id":"phi-3.5-moe-instruct","labId":"microsoft","license":"mit","name":"Phi-3.5-MoE-instruct","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/microsoft/Phi-3.5-MoE-instruct"},{"auditedOn":"2026-09-06","availability":"Downloadable Microsoft instruction/reasoning weights","contextTokens":32768,"family":"Phi","id":"phi-4-reasoning-plus","labId":"microsoft","license":"mit","name":"Phi-4-reasoning-plus","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/microsoft/Phi-4-reasoning-plus"},{"auditedOn":"2026-09-06","availability":"Downloadable Microsoft instruction/reasoning weights","contextTokens":131072,"family":"Phi","id":"phi-4-mini-reasoning","labId":"microsoft","license":"mit","name":"Phi-4-mini-reasoning","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/microsoft/Phi-4-mini-reasoning"},{"auditedOn":"2026-09-06","availability":"Downloadable Microsoft instruction/reasoning weights","contextTokens":65536,"family":"Phi","id":"phi-4-mini-flash-reasoning","labId":"microsoft","license":"mit","name":"Phi-4-mini-flash-reasoning","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/microsoft/Phi-4-mini-flash-reasoning"},{"auditedOn":"2026-09-06","availability":"Downloadable Microsoft instruction/reasoning weights","contextTokens":null,"family":"MAI","id":"mai-ds-r1","labId":"microsoft","license":"mit","name":"MAI-DS-R1","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/microsoft/MAI-DS-R1"},{"auditedOn":"2026-09-06","availability":"Microsoft Foundry public preview","contextTokens":null,"family":"MAI","id":"mai-thinking-1","labId":"microsoft","license":"proprietary","name":"MAI-Thinking-1","releasedOn":null,"retirementOn":null,"status":"preview","url":"https://microsoft.ai/models/mai-thinking-1/"},{"auditedOn":"2026-09-06","availability":"Available in GitHub Copilot and VS Code","contextTokens":null,"family":"MAI","id":"mai-code-1-1-flash","labId":"microsoft","license":"proprietary","name":"MAI-Code-1.1-Flash","releasedOn":null,"retirementOn":null,"status":"active","url":"https://microsoft.ai/models/mai-code-1-flash/"},{"auditedOn":"2026-09-06","availability":"GitHub Copilot; scheduled retirement September 10, 2026","contextTokens":null,"family":"MAI","id":"mai-code-1-flash","labId":"microsoft","license":"proprietary","name":"MAI-Code-1-Flash","releasedOn":null,"retirementOn":"2026-09-10","status":"deprecated","url":"https://docs.github.com/en/copilot/reference/ai-models/supported-models"},{"auditedOn":"2026-09-06","availability":"Public provider offering; account and region restrictions may apply","contextTokens":128000,"family":"Nova","id":"amazon-nova-micro","labId":"amazon","license":"proprietary","name":"Amazon Nova Micro","releasedOn":"2024-12-02","retirementOn":null,"status":"active","url":"https://docs.aws.amazon.com/nova/latest/userguide/additional-resources.html"},{"auditedOn":"2026-09-06","availability":"Public provider offering; account and region restrictions may apply","contextTokens":300000,"family":"Nova","id":"amazon-nova-lite","labId":"amazon","license":"proprietary","name":"Amazon Nova Lite","releasedOn":"2024-12-02","retirementOn":null,"status":"active","url":"https://docs.aws.amazon.com/nova/latest/userguide/additional-resources.html"},{"auditedOn":"2026-09-06","availability":"Public provider offering; account and region restrictions may apply","contextTokens":300000,"family":"Nova","id":"amazon-nova-pro","labId":"amazon","license":"proprietary","name":"Amazon Nova Pro","releasedOn":"2024-12-02","retirementOn":null,"status":"active","url":"https://docs.aws.amazon.com/nova/latest/userguide/additional-resources.html"},{"auditedOn":"2026-09-06","availability":"Bedrock Legacy; existing eligible customers until September 14, 2026","contextTokens":1000000,"family":"Nova","id":"amazon-nova-premier","labId":"amazon","license":"proprietary","name":"Amazon Nova Premier","releasedOn":"2025-04-30","retirementOn":"2026-09-14","status":"deprecated","url":"https://docs.aws.amazon.com/nova/latest/userguide/additional-resources.html"},{"auditedOn":"2026-09-06","availability":"Public provider offering; account and region restrictions may apply","contextTokens":1000000,"family":"Nova","id":"amazon-nova-2-lite","labId":"amazon","license":"proprietary","name":"Amazon Nova 2 Lite","releasedOn":null,"retirementOn":null,"status":"active","url":"https://docs.aws.amazon.com/nova/latest/nova2-userguide/core-inference.html"},{"auditedOn":"2026-09-06","availability":"Announced preview for Nova Forge early-access customers; current general API availability not confirmed","contextTokens":null,"family":"Nova","id":"amazon-nova-2-pro-preview","labId":"amazon","license":"proprietary","name":"Amazon Nova 2 Pro Preview","releasedOn":null,"retirementOn":null,"status":"availability-unverified","url":"https://aws.amazon.com/about-aws/whats-new/2025/12/nova-2-foundation-models-amazon-bedrock/"},{"auditedOn":"2026-09-07","availability":"Documented provider API and open weights for self-hosting","contextTokens":1000000,"family":"DeepSeek V4","id":"deepseek-v4-pro-0424","labId":"deepseek","license":"mit","name":"DeepSeek V4-Pro 0424","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://artificialanalysis.ai/models/deepseek-v4-pro-0424"},{"auditedOn":"2026-09-07","availability":"Listed in current provider API documentation; account and region restrictions may apply","contextTokens":null,"family":"Seed","id":"doubao-seed-1.6","labId":"bytedance","license":"proprietary","name":"Seed 1.6","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://www.volcengine.com/product/doubao"},{"auditedOn":"2026-09-07","availability":"Listed in current provider API documentation; account and region restrictions may apply","contextTokens":null,"family":"Seed","id":"doubao-seed-1.6-thinking","labId":"bytedance","license":"proprietary","name":"Seed 1.6 Thinking","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://www.volcengine.com/product/doubao"},{"auditedOn":"2026-09-07","availability":"Listed in current provider API documentation; account and region restrictions may apply","contextTokens":null,"family":"Seed","id":"doubao-seed-1.6-vision","labId":"bytedance","license":"proprietary","name":"Seed 1.6 Vision","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://www.volcengine.com/product/doubao"},{"auditedOn":"2026-09-07","availability":"Listed in current provider API documentation; account and region restrictions may apply","contextTokens":null,"family":"Seed","id":"doubao-seed-1.8","labId":"bytedance","license":"proprietary","name":"Seed 1.8","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://www.volcengine.com/product/doubao"},{"auditedOn":"2026-09-07","availability":"Public release documented; current endpoint access has not been verified","contextTokens":null,"family":"Seed","id":"doubao-seed-2.0-code","labId":"bytedance","license":"proprietary","name":"Seed 2.0 Code","releasedOn":"2026-02-14","retirementOn":null,"status":"availability-unverified","url":"https://seed.bytedance.com/en/blog/seed-2-0-official-launch"},{"auditedOn":"2026-09-07","availability":"Public release documented; current endpoint access has not been verified","contextTokens":null,"family":"Seed","id":"doubao-seed-2.0-lite","labId":"bytedance","license":"proprietary","name":"Seed 2.0 Lite","releasedOn":"2026-02-14","retirementOn":null,"status":"availability-unverified","url":"https://seed.bytedance.com/en/blog/seed-2-0-official-launch"},{"auditedOn":"2026-09-07","availability":"Public release documented; current endpoint access has not been verified","contextTokens":null,"family":"Seed","id":"doubao-seed-2.0-mini","labId":"bytedance","license":"proprietary","name":"Seed 2.0 Mini","releasedOn":"2026-02-14","retirementOn":null,"status":"availability-unverified","url":"https://seed.bytedance.com/en/blog/seed-2-0-official-launch"},{"auditedOn":"2026-09-07","availability":"Public release documented; current endpoint access has not been verified","contextTokens":null,"family":"Seed","id":"doubao-seed-2.0-pro","labId":"bytedance","license":"proprietary","name":"Seed 2.0 Pro","releasedOn":"2026-02-14","retirementOn":null,"status":"availability-unverified","url":"https://seed.bytedance.com/en/blog/seed-2-0-official-launch"},{"auditedOn":"2026-09-07","availability":"Listed in current provider API documentation; account and region restrictions may apply","contextTokens":null,"family":"Seed","id":"doubao-seed-2.1-pro","labId":"bytedance","license":"proprietary","name":"Seed 2.1 Pro","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://www.volcengine.com/product/doubao"},{"auditedOn":"2026-09-07","availability":"Listed in current provider API documentation; account and region restrictions may apply","contextTokens":null,"family":"Seed","id":"doubao-seed-2.1-turbo","labId":"bytedance","license":"proprietary","name":"Seed 2.1 Turbo","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://www.volcengine.com/product/doubao"},{"auditedOn":"2026-09-07","availability":"Listed in current provider API documentation; account and region restrictions may apply","contextTokens":null,"family":"Seed","id":"doubao-seed-character","labId":"bytedance","license":"proprietary","name":"Seed Character","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://www.volcengine.com/product/doubao"},{"auditedOn":"2026-09-07","availability":"Public release documented; current endpoint access has not been verified","contextTokens":null,"family":"Doubao","id":"doubao-seed-code","labId":"bytedance","license":"proprietary","name":"Doubao Seed Code","releasedOn":null,"retirementOn":null,"status":"availability-unverified","url":"https://developer.volcengine.com/articles/7577301460712030258"},{"auditedOn":"2026-09-07","availability":"Listed in current provider API documentation; account and region restrictions may apply","contextTokens":null,"family":"Seed","id":"doubao-seed-evolving","labId":"bytedance","license":"proprietary","name":"Seed Evolving","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://www.volcengine.com/product/doubao"},{"auditedOn":"2026-09-07","availability":"Public provider-owned weights available for self-hosting; API availability is not implied","contextTokens":null,"family":"Seed","id":"seed-coder-8b-instruct","labId":"bytedance","license":"mit","name":"Seed Coder 8B Instruct","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/ByteDance-Seed/Seed-Coder-8B-Instruct"},{"auditedOn":"2026-09-07","availability":"Public provider-owned weights available for self-hosting; API availability is not implied","contextTokens":null,"family":"Seed","id":"seed-coder-8b-reasoning","labId":"bytedance","license":"mit","name":"Seed Coder 8B Reasoning","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/ByteDance-Seed/Seed-Coder-8B-Reasoning"},{"auditedOn":"2026-09-07","availability":"Public provider-owned weights available for self-hosting; API availability is not implied","contextTokens":null,"family":"Seed","id":"seed-oss-36b-instruct","labId":"bytedance","license":"apache-2.0","name":"Seed OSS 36B Instruct","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/ByteDance-Seed/Seed-OSS-36B-Instruct"},{"auditedOn":"2026-09-07","availability":"Public provider-owned weights available for self-hosting; API availability is not implied","contextTokens":null,"family":"EXAONE","id":"exaone-3.0-7.8b-instruct","labId":"lgai","license":"other","name":"EXAONE 3.0 7.8B Instruct","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/LGAI-EXAONE/EXAONE-3.0-7.8B-Instruct"},{"auditedOn":"2026-09-07","availability":"Public provider-owned weights available for self-hosting; API availability is not implied","contextTokens":null,"family":"EXAONE","id":"exaone-3.5-2.4b-instruct","labId":"lgai","license":"other","name":"EXAONE 3.5 2.4B Instruct","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/LGAI-EXAONE/EXAONE-3.5-2.4B-Instruct"},{"auditedOn":"2026-09-07","availability":"Public provider-owned weights available for self-hosting; API availability is not implied","contextTokens":null,"family":"EXAONE","id":"exaone-3.5-32b-instruct","labId":"lgai","license":"other","name":"EXAONE 3.5 32B Instruct","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/LGAI-EXAONE/EXAONE-3.5-32B-Instruct"},{"auditedOn":"2026-09-07","availability":"Public provider-owned weights available for self-hosting; API availability is not implied","contextTokens":null,"family":"EXAONE","id":"exaone-3.5-7.8b-instruct","labId":"lgai","license":"other","name":"EXAONE 3.5 7.8B Instruct","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/LGAI-EXAONE/EXAONE-3.5-7.8B-Instruct"},{"auditedOn":"2026-09-07","availability":"Public provider-owned weights available for self-hosting; API availability is not implied","contextTokens":null,"family":"EXAONE","id":"exaone-4.0-1.2b","labId":"lgai","license":"other","name":"EXAONE 4.0 1.2B","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/LGAI-EXAONE/EXAONE-4.0-1.2B"},{"auditedOn":"2026-09-07","availability":"Public provider-owned weights available for self-hosting; API availability is not implied","contextTokens":null,"family":"EXAONE","id":"exaone-4.0-32b","labId":"lgai","license":"other","name":"EXAONE 4.0 32B","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/LGAI-EXAONE/EXAONE-4.0-32B"},{"auditedOn":"2026-09-07","availability":"Public provider-owned weights available for self-hosting; API availability is not implied","contextTokens":null,"family":"EXAONE","id":"exaone-4.0.1-32b","labId":"lgai","license":"other","name":"EXAONE 4.0.1 32B","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/LGAI-EXAONE/EXAONE-4.0.1-32B"},{"auditedOn":"2026-09-07","availability":"Public provider-owned weights available for self-hosting; API availability is not implied","contextTokens":null,"family":"EXAONE","id":"exaone-4.5-33b","labId":"lgai","license":"other","name":"EXAONE 4.5 33B","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/LGAI-EXAONE/EXAONE-4.5-33B"},{"auditedOn":"2026-09-07","availability":"Public provider-owned weights available for self-hosting; API availability is not implied","contextTokens":null,"family":"EXAONE","id":"exaone-deep-2.4b","labId":"lgai","license":"other","name":"EXAONE Deep 2.4B","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/LGAI-EXAONE/EXAONE-Deep-2.4B"},{"auditedOn":"2026-09-07","availability":"Public provider-owned weights available for self-hosting; API availability is not implied","contextTokens":null,"family":"EXAONE","id":"exaone-deep-32b","labId":"lgai","license":"other","name":"EXAONE Deep 32B","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/LGAI-EXAONE/EXAONE-Deep-32B"},{"auditedOn":"2026-09-07","availability":"Public provider-owned weights available for self-hosting; API availability is not implied","contextTokens":null,"family":"EXAONE","id":"exaone-deep-7.8b","labId":"lgai","license":"other","name":"EXAONE Deep 7.8B","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/LGAI-EXAONE/EXAONE-Deep-7.8B"},{"auditedOn":"2026-09-07","availability":"Public provider-owned weights available for self-hosting; API availability is not implied","contextTokens":null,"family":"K","id":"k-exaone-2.0-750b-a37b","labId":"lgai","license":"apache-2.0","name":"K EXAONE 2.0 750B A37B","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/LGAI-EXAONE/K-EXAONE-2.0-750B-A37B"},{"auditedOn":"2026-09-07","availability":"Public provider-owned weights available for self-hosting; API availability is not implied","contextTokens":null,"family":"K","id":"k-exaone-236b-a23b","labId":"lgai","license":"other","name":"K EXAONE 236B A23B","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/LGAI-EXAONE/K-EXAONE-236B-A23B"},{"auditedOn":"2026-09-07","availability":"Public provider-owned weights available for self-hosting; API availability is not implied","contextTokens":null,"family":"Step","id":"step-3.5-flash","labId":"stepfun","license":"apache-2.0","name":"Step 3.5 Flash","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/stepfun-ai/Step-3.5-Flash"},{"auditedOn":"2026-09-07","availability":"Public provider-owned weights available for self-hosting; API availability is not implied","contextTokens":null,"family":"Step","id":"step-3.7-flash","labId":"stepfun","license":"apache-2.0","name":"Step 3.7 Flash","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/stepfun-ai/Step-3.7-Flash"},{"auditedOn":"2026-09-07","availability":"Public provider-owned weights available for self-hosting; API availability is not implied","contextTokens":null,"family":"Step","id":"step3","labId":"stepfun","license":"apache-2.0","name":"Step 3","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/stepfun-ai/step3"},{"auditedOn":"2026-09-07","availability":"Public provider-owned weights available for self-hosting; API availability is not implied","contextTokens":null,"family":"Step","id":"step3-vl-10b","labId":"stepfun","license":"apache-2.0","name":"Step 3 VL 10B","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/stepfun-ai/Step3-VL-10B"},{"auditedOn":"2026-09-07","availability":"Public provider-owned weights available for self-hosting; API availability is not implied","contextTokens":null,"family":"SOLAR","id":"solar-10.7b-instruct-v1.0","labId":"upstage","license":"cc-by-nc-4.0","name":"SOLAR 10.7B Instruct v1.0","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/upstage/SOLAR-10.7B-Instruct-v1.0"},{"auditedOn":"2026-09-07","availability":"Listed in current provider API documentation; account and region restrictions may apply","contextTokens":null,"family":"Solar","id":"solar-mini-250422","labId":"upstage","license":"proprietary","name":"Solar Mini (April 2025)","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://console.upstage.ai/docs/capabilities/generate/chat"},{"auditedOn":"2026-09-07","availability":"Public provider-owned weights available for self-hosting; API availability is not implied","contextTokens":null,"family":"Solar","id":"solar-open-100b","labId":"upstage","license":"other","name":"Solar Open 100B","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/upstage/Solar-Open-100B"},{"auditedOn":"2026-09-07","availability":"Public provider-owned weights available for self-hosting; API availability is not implied","contextTokens":null,"family":"Solar","id":"solar-open2-250b","labId":"upstage","license":"other","name":"Solar Open2 250B","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/upstage/Solar-Open2-250B"},{"auditedOn":"2026-09-07","availability":"Public provider-owned weights available for self-hosting; API availability is not implied","contextTokens":null,"family":"solar","id":"solar-pro-preview-instruct","labId":"upstage","license":"mit","name":"solar pro preview instruct","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/upstage/solar-pro-preview-instruct"},{"auditedOn":"2026-09-07","availability":"Listed in current provider API documentation; account and region restrictions may apply","contextTokens":null,"family":"Solar","id":"solar-pro2-251215","labId":"upstage","license":"proprietary","name":"Solar Pro 2 (December 2025)","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://console.upstage.ai/docs/capabilities/generate/chat"},{"auditedOn":"2026-09-07","availability":"Listed in current provider API documentation; account and region restrictions may apply","contextTokens":null,"family":"Solar","id":"solar-pro3-260323","labId":"upstage","license":"proprietary","name":"Solar Pro 3","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://console.upstage.ai/docs/capabilities/generate/chat"},{"auditedOn":"2026-09-07","availability":"Listed in current provider API documentation; account and region restrictions may apply","contextTokens":null,"family":"Solar","id":"solar-pro4-260806","labId":"upstage","license":"proprietary","name":"Solar Pro 4","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://console.upstage.ai/docs/capabilities/generate/chat"},{"auditedOn":"2026-09-07","availability":"Public provider-owned weights available for self-hosting; API availability is not implied","contextTokens":null,"family":"MiMo-7B-RL","id":"mimo-7b-rl","labId":"xiaomi","license":"mit","name":"MiMo-7B-RL","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/XiaomiMiMo/MiMo-7B-RL"},{"auditedOn":"2026-09-07","availability":"Public provider-owned weights available for self-hosting; API availability is not implied","contextTokens":null,"family":"MiMo-7B-RL-0530","id":"mimo-7b-rl-0530","labId":"xiaomi","license":"mit","name":"MiMo-7B-RL-0530","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/XiaomiMiMo/MiMo-7B-RL-0530"},{"auditedOn":"2026-09-07","availability":"Public provider-owned weights available for self-hosting; API availability is not implied","contextTokens":null,"family":"MiMo-7B-RL-Zero","id":"mimo-7b-rl-zero","labId":"xiaomi","license":"mit","name":"MiMo-7B-RL-Zero","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/XiaomiMiMo/MiMo-7B-RL-Zero"},{"auditedOn":"2026-09-07","availability":"Public provider-owned weights available for self-hosting; API availability is not implied","contextTokens":null,"family":"MiMo-7B-SFT","id":"mimo-7b-sft","labId":"xiaomi","license":"mit","name":"MiMo-7B-SFT","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/XiaomiMiMo/MiMo-7B-SFT"},{"auditedOn":"2026-09-07","availability":"Public provider-owned weights available for self-hosting; API availability is not implied","contextTokens":null,"family":"MiMo-V2-Flash","id":"mimo-v2-flash","labId":"xiaomi","license":"mit","name":"MiMo-V2-Flash","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/XiaomiMiMo/MiMo-V2-Flash"},{"auditedOn":"2026-09-07","availability":"Public provider-owned weights available for self-hosting; API availability is not implied","contextTokens":null,"family":"MiMo-V2.5","id":"mimo-v2.5","labId":"xiaomi","license":"mit","name":"MiMo-V2.5","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/XiaomiMiMo/MiMo-V2.5"},{"auditedOn":"2026-09-07","availability":"Public provider-owned weights available for self-hosting; API availability is not implied","contextTokens":null,"family":"MiMo-V2.5-Pro","id":"mimo-v2.5-pro","labId":"xiaomi","license":"mit","name":"MiMo-V2.5-Pro","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/XiaomiMiMo/MiMo-V2.5-Pro"},{"auditedOn":"2026-09-07","availability":"Public provider-owned weights available for self-hosting; API availability is not implied","contextTokens":null,"family":"MiMo-VL-7B-RL","id":"mimo-vl-7b-rl","labId":"xiaomi","license":"mit","name":"MiMo-VL-7B-RL","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/XiaomiMiMo/MiMo-VL-7B-RL"},{"auditedOn":"2026-09-07","availability":"Public provider-owned weights available for self-hosting; API availability is not implied","contextTokens":null,"family":"MiMo-VL-7B-RL-2508","id":"mimo-vl-7b-rl-2508","labId":"xiaomi","license":"mit","name":"MiMo-VL-7B-RL-2508","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/XiaomiMiMo/MiMo-VL-7B-RL-2508"},{"auditedOn":"2026-09-07","availability":"Public provider-owned weights available for self-hosting; API availability is not implied","contextTokens":null,"family":"MiMo-VL-7B-SFT","id":"mimo-vl-7b-sft","labId":"xiaomi","license":"mit","name":"MiMo-VL-7B-SFT","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/XiaomiMiMo/MiMo-VL-7B-SFT"},{"auditedOn":"2026-09-07","availability":"Public provider-owned weights available for self-hosting; API availability is not implied","contextTokens":null,"family":"MiMo-VL-7B-SFT-2508","id":"mimo-vl-7b-sft-2508","labId":"xiaomi","license":"mit","name":"MiMo-VL-7B-SFT-2508","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/XiaomiMiMo/MiMo-VL-7B-SFT-2508"},{"auditedOn":"2026-09-07","availability":"Public provider-owned instruction or general multimodal weights available for self-hosting; API availability is not implied","contextTokens":null,"family":"ERNIE","id":"ernie-4.5-0.3b-pt","labId":"baidu","license":"apache-2.0","name":"ERNIE-4.5-0.3B-PT","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/baidu/ERNIE-4.5-0.3B-PT"},{"auditedOn":"2026-09-07","availability":"Public provider-owned instruction or general multimodal weights available for self-hosting; API availability is not implied","contextTokens":null,"family":"ERNIE","id":"ernie-4.5-21b-a3b-pt","labId":"baidu","license":"apache-2.0","name":"ERNIE-4.5-21B-A3B-PT","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/baidu/ERNIE-4.5-21B-A3B-PT"},{"auditedOn":"2026-09-07","availability":"Public provider-owned instruction or general multimodal weights available for self-hosting; API availability is not implied","contextTokens":null,"family":"ERNIE","id":"ernie-4.5-21b-a3b-thinking","labId":"baidu","license":"apache-2.0","name":"ERNIE-4.5-21B-A3B-Thinking","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/baidu/ERNIE-4.5-21B-A3B-Thinking"},{"auditedOn":"2026-09-07","availability":"Public provider-owned instruction or general multimodal weights available for self-hosting; API availability is not implied","contextTokens":null,"family":"ERNIE","id":"ernie-4.5-300b-a47b-pt","labId":"baidu","license":"apache-2.0","name":"ERNIE-4.5-300B-A47B-PT","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/baidu/ERNIE-4.5-300B-A47B-PT"},{"auditedOn":"2026-09-07","availability":"Public provider-owned instruction or general multimodal weights available for self-hosting; API availability is not implied","contextTokens":null,"family":"ERNIE","id":"ernie-4.5-vl-28b-a3b-pt","labId":"baidu","license":"apache-2.0","name":"ERNIE-4.5-VL-28B-A3B-PT","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/baidu/ERNIE-4.5-VL-28B-A3B-PT"},{"auditedOn":"2026-09-07","availability":"Public provider-owned instruction or general multimodal weights available for self-hosting; API availability is not implied","contextTokens":null,"family":"ERNIE","id":"ernie-4.5-vl-28b-a3b-thinking","labId":"baidu","license":"apache-2.0","name":"ERNIE-4.5-VL-28B-A3B-Thinking","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/baidu/ERNIE-4.5-VL-28B-A3B-Thinking"},{"auditedOn":"2026-09-07","availability":"Public provider-owned instruction or general multimodal weights available for self-hosting; API availability is not implied","contextTokens":null,"family":"ERNIE","id":"ernie-4.5-vl-424b-a47b-pt","labId":"baidu","license":"apache-2.0","name":"ERNIE-4.5-VL-424B-A47B-PT","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/baidu/ERNIE-4.5-VL-424B-A47B-PT"},{"auditedOn":"2026-09-07","availability":"Public provider-owned instruction or general multimodal weights available for self-hosting; API availability is not implied","contextTokens":null,"family":"Qianfan","id":"qianfan-vl-3b","labId":"baidu","license":"custom","name":"Qianfan-VL-3B","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/baidu/Qianfan-VL-3B"},{"auditedOn":"2026-09-07","availability":"Public provider-owned instruction or general multimodal weights available for self-hosting; API availability is not implied","contextTokens":null,"family":"Qianfan","id":"qianfan-vl-70b","labId":"baidu","license":"mit","name":"Qianfan-VL-70B","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/baidu/Qianfan-VL-70B"},{"auditedOn":"2026-09-07","availability":"Public provider-owned instruction or general multimodal weights available for self-hosting; API availability is not implied","contextTokens":null,"family":"Qianfan","id":"qianfan-vl-8b","labId":"baidu","license":"custom","name":"Qianfan-VL-8B","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/baidu/Qianfan-VL-8B"},{"auditedOn":"2026-09-07","availability":"Public provider-owned instruction or general multimodal weights available for self-hosting; API availability is not implied","contextTokens":null,"family":"Hunyuan","id":"hunyuan-0.5b-instruct","labId":"tencent","license":"custom","name":"Hunyuan-0.5B-Instruct","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/tencent/Hunyuan-0.5B-Instruct"},{"auditedOn":"2026-09-07","availability":"Public provider-owned instruction or general multimodal weights available for self-hosting; API availability is not implied","contextTokens":null,"family":"Hunyuan","id":"hunyuan-1.8b-instruct","labId":"tencent","license":"custom","name":"Hunyuan-1.8B-Instruct","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/tencent/Hunyuan-1.8B-Instruct"},{"auditedOn":"2026-09-07","availability":"Public provider-owned instruction or general multimodal weights available for self-hosting; API availability is not implied","contextTokens":null,"family":"Hunyuan","id":"hunyuan-4b-instruct","labId":"tencent","license":"custom","name":"Hunyuan-4B-Instruct","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/tencent/Hunyuan-4B-Instruct"},{"auditedOn":"2026-09-07","availability":"Public provider-owned instruction or general multimodal weights available for self-hosting; API availability is not implied","contextTokens":null,"family":"Hunyuan","id":"hunyuan-7b-instruct","labId":"tencent","license":"custom","name":"Hunyuan-7B-Instruct","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/tencent/Hunyuan-7B-Instruct"},{"auditedOn":"2026-09-07","availability":"Public provider-owned instruction or general multimodal weights available for self-hosting; API availability is not implied","contextTokens":null,"family":"Hunyuan","id":"hunyuan-7b-instruct-0124","labId":"tencent","license":"custom","name":"Hunyuan-7B-Instruct-0124","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/tencent/Hunyuan-7B-Instruct-0124"},{"auditedOn":"2026-09-07","availability":"Public provider-owned instruction or general multimodal weights available for self-hosting; API availability is not implied","contextTokens":262144,"family":"Hunyuan","id":"hunyuan-a13b-instruct","labId":"tencent","license":"custom","name":"Hunyuan-A13B-Instruct","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/tencent/Hunyuan-A13B-Instruct"},{"auditedOn":"2026-09-07","availability":"Public provider-owned instruction or general multimodal weights available for self-hosting; API availability is not implied","contextTokens":262144,"family":"Hy","id":"hy3","labId":"tencent","license":"apache-2.0","name":"Hy3","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/tencent/Hy3"},{"auditedOn":"2026-09-07","availability":"Public provider-owned instruction or general multimodal weights available for self-hosting; API availability is not implied","contextTokens":262144,"family":"Hy","id":"hy3-preview","labId":"tencent","license":"custom","name":"Hy3-preview","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/tencent/Hy3-preview"},{"auditedOn":"2026-09-07","availability":"Public provider-owned instruction or general multimodal weights available for self-hosting; API availability is not implied","contextTokens":1000000,"family":"Hy","id":"hy4-preview","labId":"tencent","license":"apache-2.0","name":"Hy4-preview","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/tencent/Hy4-preview"},{"auditedOn":"2026-09-07","availability":"Public provider-owned instruction or general multimodal weights available for self-hosting; API availability is not implied","contextTokens":null,"family":"Hunyuan A52B Instruct","id":"hunyuan-a52b-instruct","labId":"tencent","license":"custom","name":"Hunyuan A52B Instruct","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/tencent/Tencent-Hunyuan-Large/tree/main/Hunyuan-A52B-Instruct"},{"auditedOn":"2026-09-07","availability":"Public provider-owned instruction or general multimodal weights available for self-hosting; API availability is not implied","contextTokens":32768,"family":"WeDLM","id":"wedlm-7b-instruct","labId":"tencent","license":"apache-2.0","name":"WeDLM-7B-Instruct","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/tencent/WeDLM-7B-Instruct"},{"auditedOn":"2026-09-07","availability":"Public provider-owned instruction or general multimodal weights available for self-hosting; API availability is not implied","contextTokens":32768,"family":"WeDLM","id":"wedlm-8b-instruct","labId":"tencent","license":"apache-2.0","name":"WeDLM-8B-Instruct","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/tencent/WeDLM-8B-Instruct"},{"auditedOn":"2026-09-07","availability":"Public provider-owned instruction or general multimodal weights available for self-hosting; API availability is not implied","contextTokens":131072,"family":"Youtu","id":"youtu-llm-2b","labId":"tencent","license":"custom","name":"Youtu-LLM-2B","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/tencent/Youtu-LLM-2B"},{"auditedOn":"2026-09-07","availability":"Listed in current provider API documentation; account, region and preview restrictions may apply","contextTokens":null,"family":"ERNIE","id":"ernie-5.1","labId":"baidu","license":"proprietary","name":"ERNIE-5.1","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://cloud.baidu.com/doc/qianfan/s/rmh4stp0j"},{"auditedOn":"2026-09-07","availability":"Listed in current provider API documentation; account, region and preview restrictions may apply","contextTokens":null,"family":"ERNIE","id":"ernie-5.0","labId":"baidu","license":"proprietary","name":"ERNIE-5.0","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://cloud.baidu.com/doc/qianfan/s/rmh4stp0j"},{"auditedOn":"2026-09-07","availability":"Listed in current provider API documentation; account, region and preview restrictions may apply","contextTokens":null,"family":"ERNIE","id":"ernie-5.0-thinking-preview","labId":"baidu","license":"proprietary","name":"ERNIE-5.0-thinking-preview","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://cloud.baidu.com/doc/qianfan/s/rmh4stp0j"},{"auditedOn":"2026-09-07","availability":"Listed in current provider API documentation; account, region and preview restrictions may apply","contextTokens":null,"family":"ERNIE","id":"ernie-5.0-thinking-latest","labId":"baidu","license":"proprietary","name":"ERNIE-5.0-thinking-latest","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://cloud.baidu.com/doc/qianfan/s/rmh4stp0j"},{"auditedOn":"2026-09-07","availability":"Listed in current provider API documentation; account, region and preview restrictions may apply","contextTokens":null,"family":"ERNIE","id":"ernie-5.0-thinking-exp","labId":"baidu","license":"proprietary","name":"ERNIE-5.0-thinking-exp","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://cloud.baidu.com/doc/qianfan/s/rmh4stp0j"},{"auditedOn":"2026-09-07","availability":"Listed in current provider API documentation; account, region and preview restrictions may apply","contextTokens":null,"family":"ERNIE","id":"ernie-4.5-turbo-32k","labId":"baidu","license":"proprietary","name":"ERNIE-4.5-turbo-32k","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://cloud.baidu.com/doc/qianfan/s/rmh4stp0j"},{"auditedOn":"2026-09-07","availability":"Listed in current provider API documentation; account, region and preview restrictions may apply","contextTokens":null,"family":"ERNIE","id":"ernie-4.5-turbo-128k","labId":"baidu","license":"proprietary","name":"ERNIE-4.5-turbo-128k","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://cloud.baidu.com/doc/qianfan/s/rmh4stp0j"},{"auditedOn":"2026-09-07","availability":"Listed in current provider API documentation; account, region and preview restrictions may apply","contextTokens":null,"family":"ERNIE","id":"ernie-4.5-turbo-20260402","labId":"baidu","license":"proprietary","name":"ERNIE-4.5-turbo-20260402","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://cloud.baidu.com/doc/qianfan/s/rmh4stp0j"},{"auditedOn":"2026-09-07","availability":"Listed in current provider API documentation; account, region and preview restrictions may apply","contextTokens":null,"family":"ERNIE","id":"ernie-4.5-turbo-vl","labId":"baidu","license":"proprietary","name":"ERNIE-4.5-turbo-vl","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://cloud.baidu.com/doc/qianfan/s/rmh4stp0j"},{"auditedOn":"2026-09-07","availability":"Listed in current provider API documentation; account, region and preview restrictions may apply","contextTokens":null,"family":"ERNIE","id":"ernie-4.5-turbo-vl-32k","labId":"baidu","license":"proprietary","name":"ERNIE-4.5-turbo-vl-32k","releasedOn":null,"retirementOn":null,"status":"api-listed","url":"https://cloud.baidu.com/doc/qianfan/s/rmh4stp0j"},{"auditedOn":"2026-09-09","availability":"Downloadable instruction/reasoning weights; provider license and access terms apply","contextTokens":131072,"family":"MiniCPM5","id":"minicpm5-2b","labId":"openbmb","license":"Apache-2.0","name":"MiniCPM5-2B","releasedOn":null,"retirementOn":null,"status":"active","url":"https://huggingface.co/openbmb/MiniCPM5-2B"},{"auditedOn":"2026-09-09","availability":"Downloadable instruction/reasoning weights; provider license and access terms apply","contextTokens":1048576,"family":"Agnes","id":"agnes-2.5-pro-alpha","labId":"agnes-ai","license":"Apache-2.0","name":"Agnes 2.5 Pro Alpha","releasedOn":null,"retirementOn":null,"status":"active","url":"https://huggingface.co/Agnes-AI/Agnes-2.5-Pro-Alpha"},{"auditedOn":"2026-09-09","availability":"Downloadable instruction/reasoning weights; provider license and access terms apply","contextTokens":128000,"family":"Mistral Large","id":"mistral-large-2407","labId":"mistral","license":"Mistral Research License","name":"Mistral Large 2 (July 2024)","releasedOn":null,"retirementOn":null,"status":"active","url":"https://huggingface.co/mistralai/Mistral-Large-Instruct-2407"},{"auditedOn":"2026-09-10","availability":"Original April open weights; current API alias may point to a later checkpoint","contextTokens":1000000,"family":"DeepSeek V4","id":"deepseek-v4-flash-0424","labId":"deepseek","license":"mit","name":"DeepSeek V4 Flash 0424","releasedOn":null,"retirementOn":null,"status":"open-weights-available","url":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash"},{"auditedOn":"2026-09-28","availability":"Public provider catalog; account and region restrictions may apply","contextTokens":1000000,"family":"Claude Sonnet","id":"claude-sonnet-5-5","inputUsdPerMillion":2,"labId":"anthropic","license":"proprietary","name":"Claude Sonnet 5.5","outputUsdPerMillion":10,"releasedOn":"2026-09-28","retirementOn":null,"status":"active","url":"https://platform.claude.com/docs/en/models/sonnet-5-5/overview"}],"presets":[{"id":"default","name":"Core work","notes":"Shared core profile: agentic work, reasoning and coding. Every selected capability must be supported.","weights":{"agentic":0.34,"coding":0.33,"hard_reasoning":0.33,"human_pref":0,"knowledge":0,"long_context":0,"multimodal":0}},{"id":"coding","name":"Coding","notes":"Compare the coding capability only.","weights":{"agentic":0,"coding":1,"hard_reasoning":0,"human_pref":0,"knowledge":0,"long_context":0,"multimodal":0}},{"id":"hard_reasoning","name":"Reasoning","notes":"Compare the reasoning capability only.","weights":{"agentic":0,"coding":0,"hard_reasoning":1,"human_pref":0,"knowledge":0,"long_context":0,"multimodal":0}},{"id":"agentic","name":"Agentic","notes":"Compare the agentic capability only.","weights":{"agentic":1,"coding":0,"hard_reasoning":0,"human_pref":0,"knowledge":0,"long_context":0,"multimodal":0}},{"id":"knowledge","name":"Knowledge","notes":"Compare the knowledge capability only.","weights":{"agentic":0,"coding":0,"hard_reasoning":0,"human_pref":0,"knowledge":1,"long_context":0,"multimodal":0}},{"id":"multimodal","name":"Multimodal","notes":"Compare the multimodal capability only.","weights":{"agentic":0,"coding":0,"hard_reasoning":0,"human_pref":0,"knowledge":0,"long_context":0,"multimodal":1}},{"id":"long_context","name":"Long context","notes":"Compare the long context capability only.","weights":{"agentic":0,"coding":0,"hard_reasoning":0,"human_pref":0,"knowledge":0,"long_context":1,"multimodal":0}},{"id":"human_pref","name":"Human pref","notes":"Compare the human pref capability only.","weights":{"agentic":0,"coding":0,"hard_reasoning":0,"human_pref":1,"knowledge":0,"long_context":0,"multimodal":0}},{"id":"full","name":"All capabilities","notes":"Requires evidence for all seven capabilities. Partial profiles retain their capability estimates.","weights":{"agentic":0.14,"coding":0.16,"hard_reasoning":0.14,"human_pref":0.14,"knowledge":0.14,"long_context":0.14,"multimodal":0.14}}],"scores":[{"benchmarkId":"mmlu-pro-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"mmlu-pro-source-release-snapshot-version-not-specified-pct:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1491","ingestRunId":"launch-amazon","modelId":"amazon-nova-lite","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"MMLU-Pro","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":59,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"gpqa-diamond-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gpqa-diamond-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1492","ingestRunId":"launch-amazon","modelId":"amazon-nova-lite","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"GPQA Diamond","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":43.3,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"aime-2025-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"aime-2025-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1493","ingestRunId":"launch-amazon","modelId":"amazon-nova-lite","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"AIME","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":7,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"ifbench-prompt-loose-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"ifbench-prompt-loose-source-release-snapshot-version-not-specified-pct:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1494","ingestRunId":"launch-amazon","modelId":"amazon-nova-lite","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"IFBench prompt loose","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":34.1,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"multichallenge-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"multichallenge-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1495","ingestRunId":"launch-amazon","modelId":"amazon-nova-lite","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"MultiChallenge","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":18.7,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"mmlu-pro-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"mmlu-pro-source-release-snapshot-version-not-specified-pct:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1496","ingestRunId":"launch-amazon","modelId":"claude-haiku-4-5-20251001","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"MMLU-Pro","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":80,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"gpqa-diamond-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gpqa-diamond-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1497","ingestRunId":"launch-amazon","modelId":"claude-haiku-4-5-20251001","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"GPQA Diamond","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":73,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"aime-2025-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"aime-2025-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1498","ingestRunId":"launch-amazon","modelId":"claude-haiku-4-5-20251001","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"AIME","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":80.7,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"ifbench-prompt-loose-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"ifbench-prompt-loose-source-release-snapshot-version-not-specified-pct:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1499","ingestRunId":"launch-amazon","modelId":"claude-haiku-4-5-20251001","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"IFBench prompt loose","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":54.3,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"multichallenge-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"multichallenge-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1500","ingestRunId":"launch-amazon","modelId":"claude-haiku-4-5-20251001","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"MultiChallenge","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":57.1,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"mmlu-pro-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"mmlu-pro-source-release-snapshot-version-not-specified-pct:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1501","ingestRunId":"launch-amazon","modelId":"gpt-5-mini","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"MMLU-Pro","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":83.7,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"gpqa-diamond-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gpqa-diamond-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1502","ingestRunId":"launch-amazon","modelId":"gpt-5-mini","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"GPQA Diamond","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":82.3,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"aime-2025-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"aime-2025-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1503","ingestRunId":"launch-amazon","modelId":"gpt-5-mini","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"AIME","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":91.1,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"ifbench-prompt-loose-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"ifbench-prompt-loose-source-release-snapshot-version-not-specified-pct:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1504","ingestRunId":"launch-amazon","modelId":"gpt-5-mini","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"IFBench prompt loose","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":75.4,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"multichallenge-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"multichallenge-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1505","ingestRunId":"launch-amazon","modelId":"gpt-5-mini","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"MultiChallenge","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":67.8,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"mmlu-pro-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"mmlu-pro-source-release-snapshot-version-not-specified-pct:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1506","ingestRunId":"launch-amazon","modelId":"gemini-2.5-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"MMLU-Pro","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":83.2,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"gpqa-diamond-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gpqa-diamond-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1507","ingestRunId":"launch-amazon","modelId":"gemini-2.5-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"GPQA Diamond","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":82.8,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"aime-2025-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"aime-2025-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1508","ingestRunId":"launch-amazon","modelId":"gemini-2.5-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"AIME","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":72,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"ifbench-prompt-loose-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"ifbench-prompt-loose-source-release-snapshot-version-not-specified-pct:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1509","ingestRunId":"launch-amazon","modelId":"gemini-2.5-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"IFBench prompt loose","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":50.3,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"multichallenge-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"multichallenge-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1510","ingestRunId":"launch-amazon","modelId":"gemini-2.5-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"MultiChallenge","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":62.6,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"longcodebench-1m-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"longcodebench-1m-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1511","ingestRunId":"launch-amazon","modelId":"gemini-2.5-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"LongCodeBench 1M","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":78,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"mmlu-pro-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"mmlu-pro-source-release-snapshot-version-not-specified-pct:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1512","ingestRunId":"launch-amazon","modelId":"amazon-nova-2-lite","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"MMLU-Pro","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":80.9,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"gpqa-diamond-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gpqa-diamond-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1513","ingestRunId":"launch-amazon","modelId":"amazon-nova-2-lite","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"GPQA Diamond","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":79.6,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"aime-2025-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"aime-2025-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1514","ingestRunId":"launch-amazon","modelId":"amazon-nova-2-lite","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"AIME","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":91,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"ifbench-prompt-loose-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"ifbench-prompt-loose-source-release-snapshot-version-not-specified-pct:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1515","ingestRunId":"launch-amazon","modelId":"amazon-nova-2-lite","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"IFBench prompt loose","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":70.8,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"multichallenge-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"multichallenge-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1516","ingestRunId":"launch-amazon","modelId":"amazon-nova-2-lite","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"MultiChallenge","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":76.6,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"longcodebench-1m-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"longcodebench-1m-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1517","ingestRunId":"launch-amazon","modelId":"amazon-nova-2-lite","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"LongCodeBench 1M","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":84,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"mmlu-pro-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"mmlu-pro-source-release-snapshot-version-not-specified-pct:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1518","ingestRunId":"launch-amazon","modelId":"amazon-nova-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"MMLU-Pro","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":69.1,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"gpqa-diamond-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gpqa-diamond-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1519","ingestRunId":"launch-amazon","modelId":"amazon-nova-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"GPQA Diamond","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":49.9,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"aime-2025-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"aime-2025-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1520","ingestRunId":"launch-amazon","modelId":"amazon-nova-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"AIME","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":7,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"ifbench-prompt-loose-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"ifbench-prompt-loose-source-release-snapshot-version-not-specified-pct:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1521","ingestRunId":"launch-amazon","modelId":"amazon-nova-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"IFBench prompt loose","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":38.1,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"multichallenge-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"multichallenge-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1522","ingestRunId":"launch-amazon","modelId":"amazon-nova-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"MultiChallenge","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":26.7,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"mmlu-pro-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"mmlu-pro-source-release-snapshot-version-not-specified-pct:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1523","ingestRunId":"launch-amazon","modelId":"amazon-nova-premier","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"MMLU-Pro","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":73.3,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"gpqa-diamond-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gpqa-diamond-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1524","ingestRunId":"launch-amazon","modelId":"amazon-nova-premier","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"GPQA Diamond","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":56.9,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"aime-2025-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"aime-2025-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1525","ingestRunId":"launch-amazon","modelId":"amazon-nova-premier","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"AIME","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":17.3,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"ifbench-prompt-loose-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"ifbench-prompt-loose-source-release-snapshot-version-not-specified-pct:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1526","ingestRunId":"launch-amazon","modelId":"amazon-nova-premier","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"IFBench prompt loose","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":36.2,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"multichallenge-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"multichallenge-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1527","ingestRunId":"launch-amazon","modelId":"amazon-nova-premier","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"MultiChallenge","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":22,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"longcodebench-1m-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"longcodebench-1m-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1528","ingestRunId":"launch-amazon","modelId":"amazon-nova-premier","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"LongCodeBench 1M","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":42,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"mmlu-pro-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"mmlu-pro-source-release-snapshot-version-not-specified-pct:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1529","ingestRunId":"launch-amazon","modelId":"claude-sonnet-4-5-20250929","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"MMLU-Pro","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":87.5,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"gpqa-diamond-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gpqa-diamond-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1530","ingestRunId":"launch-amazon","modelId":"claude-sonnet-4-5-20250929","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"GPQA Diamond","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":83.4,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"aime-2025-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"aime-2025-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1531","ingestRunId":"launch-amazon","modelId":"claude-sonnet-4-5-20250929","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"AIME","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":87,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"ifbench-prompt-loose-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"ifbench-prompt-loose-source-release-snapshot-version-not-specified-pct:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1532","ingestRunId":"launch-amazon","modelId":"claude-sonnet-4-5-20250929","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"IFBench prompt loose","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":57.3,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"multichallenge-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"multichallenge-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1533","ingestRunId":"launch-amazon","modelId":"claude-sonnet-4-5-20250929","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"MultiChallenge","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":63.1,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"longcodebench-1m-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"longcodebench-1m-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1534","ingestRunId":"launch-amazon","modelId":"claude-sonnet-4-5-20250929","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"LongCodeBench 1M","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":82,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"mmlu-pro-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"mmlu-pro-source-release-snapshot-version-not-specified-pct:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1535","ingestRunId":"launch-amazon","modelId":"gpt-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"MMLU-Pro","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":87.1,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"gpqa-diamond-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gpqa-diamond-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1536","ingestRunId":"launch-amazon","modelId":"gpt-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"GPQA Diamond","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":85.7,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"aime-2025-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"aime-2025-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1537","ingestRunId":"launch-amazon","modelId":"gpt-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"AIME","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":94.6,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"ifbench-prompt-loose-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"ifbench-prompt-loose-source-release-snapshot-version-not-specified-pct:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1538","ingestRunId":"launch-amazon","modelId":"gpt-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"IFBench prompt loose","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":73.1,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"multichallenge-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"multichallenge-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1539","ingestRunId":"launch-amazon","modelId":"gpt-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"MultiChallenge","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":69.6,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"mmlu-pro-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"mmlu-pro-source-release-snapshot-version-not-specified-pct:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1540","ingestRunId":"launch-amazon","modelId":"gpt-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"MMLU-Pro","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":87,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"gpqa-diamond-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gpqa-diamond-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1541","ingestRunId":"launch-amazon","modelId":"gpt-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"GPQA Diamond","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":88.1,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"aime-2025-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"aime-2025-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1542","ingestRunId":"launch-amazon","modelId":"gpt-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"AIME","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":94,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"ifbench-prompt-loose-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"ifbench-prompt-loose-source-release-snapshot-version-not-specified-pct:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1543","ingestRunId":"launch-amazon","modelId":"gpt-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"IFBench prompt loose","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":72.9,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"multichallenge-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"multichallenge-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1544","ingestRunId":"launch-amazon","modelId":"gpt-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"MultiChallenge","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":72.2,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"mmlu-pro-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"mmlu-pro-source-release-snapshot-version-not-specified-pct:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1545","ingestRunId":"launch-amazon","modelId":"gemini-2.5-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"MMLU-Pro","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":86.2,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"gpqa-diamond-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gpqa-diamond-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1546","ingestRunId":"launch-amazon","modelId":"gemini-2.5-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"GPQA Diamond","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":86.4,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"aime-2025-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"aime-2025-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1547","ingestRunId":"launch-amazon","modelId":"gemini-2.5-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"AIME","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":88,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"ifbench-prompt-loose-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"ifbench-prompt-loose-source-release-snapshot-version-not-specified-pct:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1548","ingestRunId":"launch-amazon","modelId":"gemini-2.5-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"IFBench prompt loose","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":48.7,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"multichallenge-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"multichallenge-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1549","ingestRunId":"launch-amazon","modelId":"gemini-2.5-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"MultiChallenge","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":62.3,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"longcodebench-1m-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"longcodebench-1m-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1550","ingestRunId":"launch-amazon","modelId":"gemini-2.5-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"LongCodeBench 1M","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":74,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"mmlu-pro-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"mmlu-pro-source-release-snapshot-version-not-specified-pct:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1551","ingestRunId":"launch-amazon","modelId":"amazon-nova-2-pro-preview","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"MMLU-Pro","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":81.6,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"gpqa-diamond-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gpqa-diamond-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1552","ingestRunId":"launch-amazon","modelId":"amazon-nova-2-pro-preview","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"GPQA Diamond","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":81.4,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"aime-2025-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"aime-2025-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1553","ingestRunId":"launch-amazon","modelId":"amazon-nova-2-pro-preview","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"AIME","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":92.3,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"ifbench-prompt-loose-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"ifbench-prompt-loose-source-release-snapshot-version-not-specified-pct:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1554","ingestRunId":"launch-amazon","modelId":"amazon-nova-2-pro-preview","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"IFBench prompt loose","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":80.2,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"multichallenge-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"multichallenge-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1555","ingestRunId":"launch-amazon","modelId":"amazon-nova-2-pro-preview","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"MultiChallenge","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":77.7,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"longcodebench-1m-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"longcodebench-1m-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1556","ingestRunId":"launch-amazon","modelId":"amazon-nova-2-pro-preview","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"LongCodeBench 1M","locator":"Table1","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":84,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"tau2-telecom-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"tau2-telecom-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1557","ingestRunId":"launch-amazon","modelId":"amazon-nova-lite","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"tau2 Telecom","locator":"Table2","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":17.5,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"bfcl-v4-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"bfcl-v4-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1558","ingestRunId":"launch-amazon","modelId":"amazon-nova-lite","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"BFCL v4","locator":"Table2","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":33.2,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"tau2-telecom-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"tau2-telecom-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1559","ingestRunId":"launch-amazon","modelId":"claude-haiku-4-5-20251001","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"tau2 Telecom","locator":"Table2","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":54.7,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"tau2-retail-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"tau2-retail-verified-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1560","ingestRunId":"launch-amazon","modelId":"claude-haiku-4-5-20251001","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"tau2 Retail Verified","locator":"Table2","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":69.1,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"tau2-airline-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"tau2-airline-verified-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1561","ingestRunId":"launch-amazon","modelId":"claude-haiku-4-5-20251001","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"tau2 Airline Verified","locator":"Table2","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":54,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"bfcl-v4-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"bfcl-v4-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1562","ingestRunId":"launch-amazon","modelId":"claude-haiku-4-5-20251001","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"BFCL v4","locator":"Table2","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":61.8,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"tau2-telecom-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"tau2-telecom-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1563","ingestRunId":"launch-amazon","modelId":"gpt-5-mini","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"tau2 Telecom","locator":"Table2","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":71.1,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"tau2-retail-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"tau2-retail-verified-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1564","ingestRunId":"launch-amazon","modelId":"gpt-5-mini","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"tau2 Retail Verified","locator":"Table2","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":73.7,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"tau2-airline-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"tau2-airline-verified-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1565","ingestRunId":"launch-amazon","modelId":"gpt-5-mini","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"tau2 Airline Verified","locator":"Table2","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":68.8,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"bfcl-v4-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"bfcl-v4-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1566","ingestRunId":"launch-amazon","modelId":"gpt-5-mini","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"BFCL v4","locator":"Table2","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":56,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"mcpatlas-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mcpatlas-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1567","ingestRunId":"launch-amazon","modelId":"gpt-5-mini","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"MCPAtlas","locator":"Table2","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":22.2,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"tau2-telecom-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"tau2-telecom-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1568","ingestRunId":"launch-amazon","modelId":"gemini-2.5-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"tau2 Telecom","locator":"Table2","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":31.6,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"tau2-retail-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"tau2-retail-verified-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1569","ingestRunId":"launch-amazon","modelId":"gemini-2.5-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"tau2 Retail Verified","locator":"Table2","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":57.7,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"tau2-airline-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"tau2-airline-verified-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1570","ingestRunId":"launch-amazon","modelId":"gemini-2.5-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"tau2 Airline Verified","locator":"Table2","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":44,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"bfcl-v4-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"bfcl-v4-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1571","ingestRunId":"launch-amazon","modelId":"gemini-2.5-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"BFCL v4","locator":"Table2","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":55.9,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"tau2-telecom-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"tau2-telecom-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1572","ingestRunId":"launch-amazon","modelId":"amazon-nova-2-lite","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"tau2 Telecom","locator":"Table2","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":76,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"tau2-retail-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"tau2-retail-verified-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1573","ingestRunId":"launch-amazon","modelId":"amazon-nova-2-lite","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"tau2 Retail Verified","locator":"Table2","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":76.5,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"tau2-airline-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"tau2-airline-verified-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1574","ingestRunId":"launch-amazon","modelId":"amazon-nova-2-lite","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"tau2 Airline Verified","locator":"Table2","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":64.8,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"bfcl-v4-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"bfcl-v4-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1575","ingestRunId":"launch-amazon","modelId":"amazon-nova-2-lite","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"BFCL v4","locator":"Table2","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":60.3,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"mcpatlas-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mcpatlas-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1576","ingestRunId":"launch-amazon","modelId":"amazon-nova-2-lite","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"MCPAtlas","locator":"Table2","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":24.6,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"tau2-telecom-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"tau2-telecom-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1577","ingestRunId":"launch-amazon","modelId":"amazon-nova-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"tau2 Telecom","locator":"Table2","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":14,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"bfcl-v4-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"bfcl-v4-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1578","ingestRunId":"launch-amazon","modelId":"amazon-nova-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"BFCL v4","locator":"Table2","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":38,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"tau2-telecom-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"tau2-telecom-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1579","ingestRunId":"launch-amazon","modelId":"amazon-nova-premier","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"tau2 Telecom","locator":"Table2","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":38.3,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"bfcl-v4-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"bfcl-v4-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1580","ingestRunId":"launch-amazon","modelId":"amazon-nova-premier","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"BFCL v4","locator":"Table2","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":53.9,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"tau2-telecom-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"tau2-telecom-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1581","ingestRunId":"launch-amazon","modelId":"claude-sonnet-4-5-20250929","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"tau2 Telecom","locator":"Table2","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":78.1,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"tau2-retail-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"tau2-retail-verified-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1582","ingestRunId":"launch-amazon","modelId":"claude-sonnet-4-5-20250929","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"tau2 Retail Verified","locator":"Table2","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":77.2,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"tau2-airline-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"tau2-airline-verified-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1583","ingestRunId":"launch-amazon","modelId":"claude-sonnet-4-5-20250929","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"tau2 Airline Verified","locator":"Table2","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":66.8,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"bfcl-v4-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"bfcl-v4-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1584","ingestRunId":"launch-amazon","modelId":"claude-sonnet-4-5-20250929","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"BFCL v4","locator":"Table2","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":68.7,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"mcpatlas-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mcpatlas-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1585","ingestRunId":"launch-amazon","modelId":"claude-sonnet-4-5-20250929","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"MCPAtlas","locator":"Table2","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":43.8,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"tau2-telecom-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"tau2-telecom-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1586","ingestRunId":"launch-amazon","modelId":"gpt-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"tau2 Telecom","locator":"Table2","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":86.5,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"tau2-retail-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"tau2-retail-verified-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1587","ingestRunId":"launch-amazon","modelId":"gpt-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"tau2 Retail Verified","locator":"Table2","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":78.3,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"tau2-airline-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"tau2-airline-verified-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1588","ingestRunId":"launch-amazon","modelId":"gpt-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"tau2 Airline Verified","locator":"Table2","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":72,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"bfcl-v4-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"bfcl-v4-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1589","ingestRunId":"launch-amazon","modelId":"gpt-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"BFCL v4","locator":"Table2","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":61.6,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"mcpatlas-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mcpatlas-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1590","ingestRunId":"launch-amazon","modelId":"gpt-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"MCPAtlas","locator":"Table2","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":44.5,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"tau2-telecom-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"tau2-telecom-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1591","ingestRunId":"launch-amazon","modelId":"gpt-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"tau2 Telecom","locator":"Table2","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":81.9,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"tau2-retail-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"tau2-retail-verified-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1592","ingestRunId":"launch-amazon","modelId":"gpt-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"tau2 Retail Verified","locator":"Table2","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":77.5,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"tau2-airline-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"tau2-airline-verified-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1593","ingestRunId":"launch-amazon","modelId":"gpt-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"tau2 Airline Verified","locator":"Table2","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":72.4,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"bfcl-v4-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"bfcl-v4-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1594","ingestRunId":"launch-amazon","modelId":"gpt-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"BFCL v4","locator":"Table2","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":60.1,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"tau2-telecom-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"tau2-telecom-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1595","ingestRunId":"launch-amazon","modelId":"gemini-2.5-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"tau2 Telecom","locator":"Table2","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":54.1,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"tau2-retail-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"tau2-retail-verified-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1596","ingestRunId":"launch-amazon","modelId":"gemini-2.5-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"tau2 Retail Verified","locator":"Table2","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":71.3,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"tau2-airline-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"tau2-airline-verified-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1597","ingestRunId":"launch-amazon","modelId":"gemini-2.5-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"tau2 Airline Verified","locator":"Table2","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":60,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"bfcl-v4-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"bfcl-v4-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1598","ingestRunId":"launch-amazon","modelId":"gemini-2.5-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"BFCL v4","locator":"Table2","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":52.3,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"mcpatlas-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mcpatlas-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1599","ingestRunId":"launch-amazon","modelId":"gemini-2.5-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"MCPAtlas","locator":"Table2","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":8.8,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"tau2-telecom-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"tau2-telecom-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1600","ingestRunId":"launch-amazon","modelId":"amazon-nova-2-pro-preview","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"tau2 Telecom","locator":"Table2","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":92.7,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"tau2-retail-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"tau2-retail-verified-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1601","ingestRunId":"launch-amazon","modelId":"amazon-nova-2-pro-preview","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"tau2 Retail Verified","locator":"Table2","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":77.7,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"tau2-airline-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"tau2-airline-verified-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1602","ingestRunId":"launch-amazon","modelId":"amazon-nova-2-pro-preview","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"tau2 Airline Verified","locator":"Table2","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":65.2,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"bfcl-v4-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"bfcl-v4-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1603","ingestRunId":"launch-amazon","modelId":"amazon-nova-2-pro-preview","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"BFCL v4","locator":"Table2","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":61.6,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"mmmu-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mmmu-pro-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1604","ingestRunId":"launch-amazon","modelId":"amazon-nova-lite","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"MMMU-Pro","locator":"Table3","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":38,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"ocrbench-v2-average-accuracy-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"ocrbench-v2-average-accuracy-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1605","ingestRunId":"launch-amazon","modelId":"amazon-nova-lite","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"OCRBench v2 average accuracy","locator":"Table3","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":55.1,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"realkie-fcc-verified-alns-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"realkie-fcc-verified-alns-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1606","ingestRunId":"launch-amazon","modelId":"amazon-nova-lite","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"RealKIE-FCC Verified ALNS","locator":"Table3","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":40.4,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"qvhighlights-r1-0-5","evidenceKind":"lab_self_report","harnessId":"qvhighlights-r1-0-5:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1607","ingestRunId":"launch-amazon","modelId":"amazon-nova-lite","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"QVHighlights R1@0.5","locator":"Table3","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":11.9,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"screenspot-point-accuracy-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"screenspot-point-accuracy-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1608","ingestRunId":"launch-amazon","modelId":"amazon-nova-lite","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"ScreenSpot point accuracy","locator":"Table3","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":80.1,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"mmmu-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mmmu-pro-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1609","ingestRunId":"launch-amazon","modelId":"claude-haiku-4-5-20251001","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"MMMU-Pro","locator":"Table3","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":58,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"ocrbench-v2-average-accuracy-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"ocrbench-v2-average-accuracy-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1610","ingestRunId":"launch-amazon","modelId":"claude-haiku-4-5-20251001","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"OCRBench v2 average accuracy","locator":"Table3","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":48.3,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"realkie-fcc-verified-alns-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"realkie-fcc-verified-alns-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1611","ingestRunId":"launch-amazon","modelId":"claude-haiku-4-5-20251001","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"RealKIE-FCC Verified ALNS","locator":"Table3","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":52.2,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"mmmu-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mmmu-pro-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1612","ingestRunId":"launch-amazon","modelId":"gpt-5-mini","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"MMMU-Pro","locator":"Table3","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":74.1,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"ocrbench-v2-average-accuracy-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"ocrbench-v2-average-accuracy-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1613","ingestRunId":"launch-amazon","modelId":"gpt-5-mini","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"OCRBench v2 average accuracy","locator":"Table3","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":55.4,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"realkie-fcc-verified-alns-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"realkie-fcc-verified-alns-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1614","ingestRunId":"launch-amazon","modelId":"gpt-5-mini","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"RealKIE-FCC Verified ALNS","locator":"Table3","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":53.1,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"screenspot-point-accuracy-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"screenspot-point-accuracy-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1615","ingestRunId":"launch-amazon","modelId":"gpt-5-mini","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"ScreenSpot point accuracy","locator":"Table3","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":24.8,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"mmmu-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mmmu-pro-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1616","ingestRunId":"launch-amazon","modelId":"gemini-2.5-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"MMMU-Pro","locator":"Table3","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":69,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"ocrbench-v2-average-accuracy-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"ocrbench-v2-average-accuracy-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1617","ingestRunId":"launch-amazon","modelId":"gemini-2.5-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"OCRBench v2 average accuracy","locator":"Table3","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":58.2,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"realkie-fcc-verified-alns-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"realkie-fcc-verified-alns-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1618","ingestRunId":"launch-amazon","modelId":"gemini-2.5-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"RealKIE-FCC Verified ALNS","locator":"Table3","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":50.1,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"qvhighlights-r1-0-5","evidenceKind":"lab_self_report","harnessId":"qvhighlights-r1-0-5:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1619","ingestRunId":"launch-amazon","modelId":"gemini-2.5-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"QVHighlights R1@0.5","locator":"Table3","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":52.4,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"screenspot-point-accuracy-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"screenspot-point-accuracy-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1620","ingestRunId":"launch-amazon","modelId":"gemini-2.5-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"ScreenSpot point accuracy","locator":"Table3","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":27.2,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"mmmu-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mmmu-pro-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1621","ingestRunId":"launch-amazon","modelId":"amazon-nova-2-lite","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"MMMU-Pro","locator":"Table3","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":61.8,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"ocrbench-v2-average-accuracy-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"ocrbench-v2-average-accuracy-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1622","ingestRunId":"launch-amazon","modelId":"amazon-nova-2-lite","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"OCRBench v2 average accuracy","locator":"Table3","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":56.1,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"realkie-fcc-verified-alns-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"realkie-fcc-verified-alns-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1623","ingestRunId":"launch-amazon","modelId":"amazon-nova-2-lite","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"RealKIE-FCC Verified ALNS","locator":"Table3","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":62.1,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"qvhighlights-r1-0-5","evidenceKind":"lab_self_report","harnessId":"qvhighlights-r1-0-5:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1624","ingestRunId":"launch-amazon","modelId":"amazon-nova-2-lite","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"QVHighlights R1@0.5","locator":"Table3","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":77.2,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"screenspot-point-accuracy-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"screenspot-point-accuracy-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1625","ingestRunId":"launch-amazon","modelId":"amazon-nova-2-lite","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"ScreenSpot point accuracy","locator":"Table3","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":83.3,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"mmmu-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mmmu-pro-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1626","ingestRunId":"launch-amazon","modelId":"amazon-nova-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"MMMU-Pro","locator":"Table3","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":44,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"ocrbench-v2-average-accuracy-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"ocrbench-v2-average-accuracy-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1627","ingestRunId":"launch-amazon","modelId":"amazon-nova-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"OCRBench v2 average accuracy","locator":"Table3","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":59.6,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"realkie-fcc-verified-alns-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"realkie-fcc-verified-alns-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1628","ingestRunId":"launch-amazon","modelId":"amazon-nova-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"RealKIE-FCC Verified ALNS","locator":"Table3","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":49.7,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"qvhighlights-r1-0-5","evidenceKind":"lab_self_report","harnessId":"qvhighlights-r1-0-5:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1629","ingestRunId":"launch-amazon","modelId":"amazon-nova-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"QVHighlights R1@0.5","locator":"Table3","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":10.7,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"screenspot-point-accuracy-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"screenspot-point-accuracy-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1630","ingestRunId":"launch-amazon","modelId":"amazon-nova-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"ScreenSpot point accuracy","locator":"Table3","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":83.5,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"mmmu-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mmmu-pro-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1631","ingestRunId":"launch-amazon","modelId":"amazon-nova-premier","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"MMMU-Pro","locator":"Table3","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":52.4,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"ocrbench-v2-average-accuracy-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"ocrbench-v2-average-accuracy-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1632","ingestRunId":"launch-amazon","modelId":"amazon-nova-premier","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"OCRBench v2 average accuracy","locator":"Table3","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":61.2,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"realkie-fcc-verified-alns-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"realkie-fcc-verified-alns-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1633","ingestRunId":"launch-amazon","modelId":"amazon-nova-premier","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"RealKIE-FCC Verified ALNS","locator":"Table3","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":54.6,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"qvhighlights-r1-0-5","evidenceKind":"lab_self_report","harnessId":"qvhighlights-r1-0-5:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1634","ingestRunId":"launch-amazon","modelId":"amazon-nova-premier","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"QVHighlights R1@0.5","locator":"Table3","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":4.5,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"screenspot-point-accuracy-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"screenspot-point-accuracy-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1635","ingestRunId":"launch-amazon","modelId":"amazon-nova-premier","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"ScreenSpot point accuracy","locator":"Table3","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":86.2,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"mmmu-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mmmu-pro-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1636","ingestRunId":"launch-amazon","modelId":"claude-sonnet-4-5-20250929","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"MMMU-Pro","locator":"Table3","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":69,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"ocrbench-v2-average-accuracy-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"ocrbench-v2-average-accuracy-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1637","ingestRunId":"launch-amazon","modelId":"claude-sonnet-4-5-20250929","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"OCRBench v2 average accuracy","locator":"Table3","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":52.3,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"realkie-fcc-verified-alns-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"realkie-fcc-verified-alns-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1638","ingestRunId":"launch-amazon","modelId":"claude-sonnet-4-5-20250929","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"RealKIE-FCC Verified ALNS","locator":"Table3","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":62.4,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"mmmu-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mmmu-pro-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1639","ingestRunId":"launch-amazon","modelId":"gpt-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"MMMU-Pro","locator":"Table3","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":78.4,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"ocrbench-v2-average-accuracy-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"ocrbench-v2-average-accuracy-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1640","ingestRunId":"launch-amazon","modelId":"gpt-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"OCRBench v2 average accuracy","locator":"Table3","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":58.1,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"realkie-fcc-verified-alns-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"realkie-fcc-verified-alns-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1641","ingestRunId":"launch-amazon","modelId":"gpt-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"RealKIE-FCC Verified ALNS","locator":"Table3","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":51.9,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"screenspot-point-accuracy-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"screenspot-point-accuracy-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1642","ingestRunId":"launch-amazon","modelId":"gpt-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"ScreenSpot point accuracy","locator":"Table3","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":31,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"mmmu-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mmmu-pro-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1643","ingestRunId":"launch-amazon","modelId":"gpt-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"MMMU-Pro","locator":"Table3","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":76,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"ocrbench-v2-average-accuracy-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"ocrbench-v2-average-accuracy-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1644","ingestRunId":"launch-amazon","modelId":"gpt-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"OCRBench v2 average accuracy","locator":"Table3","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":57.2,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"realkie-fcc-verified-alns-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"realkie-fcc-verified-alns-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1645","ingestRunId":"launch-amazon","modelId":"gpt-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"RealKIE-FCC Verified ALNS","locator":"Table3","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":53.6,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"screenspot-point-accuracy-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"screenspot-point-accuracy-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1646","ingestRunId":"launch-amazon","modelId":"gpt-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"ScreenSpot point accuracy","locator":"Table3","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":28.5,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"mmmu-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mmmu-pro-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1647","ingestRunId":"launch-amazon","modelId":"gemini-2.5-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"MMMU-Pro","locator":"Table3","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":68,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"ocrbench-v2-average-accuracy-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"ocrbench-v2-average-accuracy-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1648","ingestRunId":"launch-amazon","modelId":"gemini-2.5-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"OCRBench v2 average accuracy","locator":"Table3","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":59.3,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"realkie-fcc-verified-alns-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"realkie-fcc-verified-alns-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1649","ingestRunId":"launch-amazon","modelId":"gemini-2.5-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"RealKIE-FCC Verified ALNS","locator":"Table3","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":54.9,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"qvhighlights-r1-0-5","evidenceKind":"lab_self_report","harnessId":"qvhighlights-r1-0-5:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1650","ingestRunId":"launch-amazon","modelId":"gemini-2.5-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"QVHighlights R1@0.5","locator":"Table3","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":75,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"screenspot-point-accuracy-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"screenspot-point-accuracy-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1651","ingestRunId":"launch-amazon","modelId":"gemini-2.5-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"ScreenSpot point accuracy","locator":"Table3","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":44.7,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"mmmu-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mmmu-pro-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1652","ingestRunId":"launch-amazon","modelId":"amazon-nova-2-pro-preview","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"MMMU-Pro","locator":"Table3","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":63.5,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"ocrbench-v2-average-accuracy-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"ocrbench-v2-average-accuracy-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1653","ingestRunId":"launch-amazon","modelId":"amazon-nova-2-pro-preview","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"OCRBench v2 average accuracy","locator":"Table3","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":64.5,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"realkie-fcc-verified-alns-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"realkie-fcc-verified-alns-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1654","ingestRunId":"launch-amazon","modelId":"amazon-nova-2-pro-preview","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"RealKIE-FCC Verified ALNS","locator":"Table3","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":67,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"qvhighlights-r1-0-5","evidenceKind":"lab_self_report","harnessId":"qvhighlights-r1-0-5:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1655","ingestRunId":"launch-amazon","modelId":"amazon-nova-2-pro-preview","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"QVHighlights R1@0.5","locator":"Table3","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":76.7,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"screenspot-point-accuracy-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"screenspot-point-accuracy-source-release-snapshot-version-not-specified:amazon:Tm92YTIgbGF1bmNoIGV2YWx1YXRpb247IGJlbmNobWFyay1zcGVjaWZpYyBzZXR0aW5ncyBpbiByZXBvcnQuIHRhdTIgVmVyaWZpZWQgYXZnQDM7IHRlbGVjb20gY2l0ZWQgQUEgYmVjYXVzZSB1c2VyIG1vZGVsIHNlbnNpdGl2aXR5OyBjb21wYXJhdG9ycyBtaXggb3duIHJ1bnMgYW5kIGNpdGVkIHByb3ZpZGVyIHNjb3Jlcy4","id":"launch-amazon-1656","ingestRunId":"launch-amazon","modelId":"amazon-nova-2-pro-preview","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores.","family":"ScreenSpot point accuracy","locator":"Table3","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":88.1,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct:amazon:Tm92YTIgaW50ZXJuYWwgYWdlbnRpYyBzY2FmZm9sZDsgc2NhbGVkIGluZmVyZW5jZSBleHBsaWNpdGx5IHNlcGFyYXRlLiBDb21wYXJhdG9yIHNldHVwcyBmcm9tIGNpdGVkIHByb3ZpZGVyIHJlcG9ydC4","id":"launch-amazon-1657","ingestRunId":"launch-amazon","modelId":"claude-haiku-4-5-20251001","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report.","family":"SWE-bench Verified","locator":"Table4","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":73.3,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"terminal-bench-1-0","evidenceKind":"lab_self_report","harnessId":"terminal-bench-1-0:amazon:Tm92YTIgaW50ZXJuYWwgYWdlbnRpYyBzY2FmZm9sZDsgc2NhbGVkIGluZmVyZW5jZSBleHBsaWNpdGx5IHNlcGFyYXRlLiBDb21wYXJhdG9yIHNldHVwcyBmcm9tIGNpdGVkIHByb3ZpZGVyIHJlcG9ydC4","id":"launch-amazon-1658","ingestRunId":"launch-amazon","modelId":"claude-haiku-4-5-20251001","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report.","family":"Terminal-Bench","locator":"Table4","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":41.8,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"livecodebench-v5-july2024-january2025-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"livecodebench-v5-july2024-january2025-source-release-snapshot-version-not-specified:amazon:Tm92YTIgaW50ZXJuYWwgYWdlbnRpYyBzY2FmZm9sZDsgc2NhbGVkIGluZmVyZW5jZSBleHBsaWNpdGx5IHNlcGFyYXRlLiBDb21wYXJhdG9yIHNldHVwcyBmcm9tIGNpdGVkIHByb3ZpZGVyIHJlcG9ydC4","id":"launch-amazon-1659","ingestRunId":"launch-amazon","modelId":"claude-haiku-4-5-20251001","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report.","family":"LiveCodeBench","locator":"Table4","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":61.5,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct:amazon:Tm92YTIgaW50ZXJuYWwgYWdlbnRpYyBzY2FmZm9sZDsgc2NhbGVkIGluZmVyZW5jZSBleHBsaWNpdGx5IHNlcGFyYXRlLiBDb21wYXJhdG9yIHNldHVwcyBmcm9tIGNpdGVkIHByb3ZpZGVyIHJlcG9ydC4","id":"launch-amazon-1660","ingestRunId":"launch-amazon","modelId":"gpt-5-mini","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report.","family":"SWE-bench Verified","locator":"Table4","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":71,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"terminal-bench-1-0","evidenceKind":"lab_self_report","harnessId":"terminal-bench-1-0:amazon:Tm92YTIgaW50ZXJuYWwgYWdlbnRpYyBzY2FmZm9sZDsgc2NhbGVkIGluZmVyZW5jZSBleHBsaWNpdGx5IHNlcGFyYXRlLiBDb21wYXJhdG9yIHNldHVwcyBmcm9tIGNpdGVkIHByb3ZpZGVyIHJlcG9ydC4","id":"launch-amazon-1661","ingestRunId":"launch-amazon","modelId":"gpt-5-mini","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report.","family":"Terminal-Bench","locator":"Table4","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":30.8,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"livecodebench-v5-july2024-january2025-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"livecodebench-v5-july2024-january2025-source-release-snapshot-version-not-specified:amazon:Tm92YTIgaW50ZXJuYWwgYWdlbnRpYyBzY2FmZm9sZDsgc2NhbGVkIGluZmVyZW5jZSBleHBsaWNpdGx5IHNlcGFyYXRlLiBDb21wYXJhdG9yIHNldHVwcyBmcm9tIGNpdGVkIHByb3ZpZGVyIHJlcG9ydC4","id":"launch-amazon-1662","ingestRunId":"launch-amazon","modelId":"gpt-5-mini","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report.","family":"LiveCodeBench","locator":"Table4","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":83.8,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct:amazon:Tm92YTIgaW50ZXJuYWwgYWdlbnRpYyBzY2FmZm9sZDsgc2NhbGVkIGluZmVyZW5jZSBleHBsaWNpdGx5IHNlcGFyYXRlLiBDb21wYXJhdG9yIHNldHVwcyBmcm9tIGNpdGVkIHByb3ZpZGVyIHJlcG9ydC4","id":"launch-amazon-1663","ingestRunId":"launch-amazon","modelId":"gemini-2.5-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report.","family":"SWE-bench Verified","locator":"Table4","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":48.9,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"swe-bench-verified-scaled-inference-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"swe-bench-verified-scaled-inference-source-release-snapshot-version-not-specified:amazon:Tm92YTIgaW50ZXJuYWwgYWdlbnRpYyBzY2FmZm9sZDsgc2NhbGVkIGluZmVyZW5jZSBleHBsaWNpdGx5IHNlcGFyYXRlLiBDb21wYXJhdG9yIHNldHVwcyBmcm9tIGNpdGVkIHByb3ZpZGVyIHJlcG9ydC4","id":"launch-amazon-1664","ingestRunId":"launch-amazon","modelId":"gemini-2.5-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report.","family":"SWE-bench Verified","locator":"Table4","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":60.4,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"terminal-bench-1-0","evidenceKind":"lab_self_report","harnessId":"terminal-bench-1-0:amazon:Tm92YTIgaW50ZXJuYWwgYWdlbnRpYyBzY2FmZm9sZDsgc2NhbGVkIGluZmVyZW5jZSBleHBsaWNpdGx5IHNlcGFyYXRlLiBDb21wYXJhdG9yIHNldHVwcyBmcm9tIGNpdGVkIHByb3ZpZGVyIHJlcG9ydC4","id":"launch-amazon-1665","ingestRunId":"launch-amazon","modelId":"gemini-2.5-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report.","family":"Terminal-Bench","locator":"Table4","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":16.8,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"livecodebench-v5-july2024-january2025-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"livecodebench-v5-july2024-january2025-source-release-snapshot-version-not-specified:amazon:Tm92YTIgaW50ZXJuYWwgYWdlbnRpYyBzY2FmZm9sZDsgc2NhbGVkIGluZmVyZW5jZSBleHBsaWNpdGx5IHNlcGFyYXRlLiBDb21wYXJhdG9yIHNldHVwcyBmcm9tIGNpdGVkIHByb3ZpZGVyIHJlcG9ydC4","id":"launch-amazon-1666","ingestRunId":"launch-amazon","modelId":"gemini-2.5-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report.","family":"LiveCodeBench","locator":"Table4","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":69.5,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct:amazon:Tm92YTIgaW50ZXJuYWwgYWdlbnRpYyBzY2FmZm9sZDsgc2NhbGVkIGluZmVyZW5jZSBleHBsaWNpdGx5IHNlcGFyYXRlLiBDb21wYXJhdG9yIHNldHVwcyBmcm9tIGNpdGVkIHByb3ZpZGVyIHJlcG9ydC4","id":"launch-amazon-1667","ingestRunId":"launch-amazon","modelId":"amazon-nova-2-lite","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report.","family":"SWE-bench Verified","locator":"Table4","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":53.6,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"swe-bench-verified-scaled-inference-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"swe-bench-verified-scaled-inference-source-release-snapshot-version-not-specified:amazon:Tm92YTIgaW50ZXJuYWwgYWdlbnRpYyBzY2FmZm9sZDsgc2NhbGVkIGluZmVyZW5jZSBleHBsaWNpdGx5IHNlcGFyYXRlLiBDb21wYXJhdG9yIHNldHVwcyBmcm9tIGNpdGVkIHByb3ZpZGVyIHJlcG9ydC4","id":"launch-amazon-1668","ingestRunId":"launch-amazon","modelId":"amazon-nova-2-lite","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report.","family":"SWE-bench Verified","locator":"Table4","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":64.5,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"terminal-bench-1-0","evidenceKind":"lab_self_report","harnessId":"terminal-bench-1-0:amazon:Tm92YTIgaW50ZXJuYWwgYWdlbnRpYyBzY2FmZm9sZDsgc2NhbGVkIGluZmVyZW5jZSBleHBsaWNpdGx5IHNlcGFyYXRlLiBDb21wYXJhdG9yIHNldHVwcyBmcm9tIGNpdGVkIHByb3ZpZGVyIHJlcG9ydC4","id":"launch-amazon-1669","ingestRunId":"launch-amazon","modelId":"amazon-nova-2-lite","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report.","family":"Terminal-Bench","locator":"Table4","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":32.5,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"livecodebench-v5-july2024-january2025-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"livecodebench-v5-july2024-january2025-source-release-snapshot-version-not-specified:amazon:Tm92YTIgaW50ZXJuYWwgYWdlbnRpYyBzY2FmZm9sZDsgc2NhbGVkIGluZmVyZW5jZSBleHBsaWNpdGx5IHNlcGFyYXRlLiBDb21wYXJhdG9yIHNldHVwcyBmcm9tIGNpdGVkIHByb3ZpZGVyIHJlcG9ydC4","id":"launch-amazon-1670","ingestRunId":"launch-amazon","modelId":"amazon-nova-2-lite","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report.","family":"LiveCodeBench","locator":"Table4","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":71,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct:amazon:Tm92YTIgaW50ZXJuYWwgYWdlbnRpYyBzY2FmZm9sZDsgc2NhbGVkIGluZmVyZW5jZSBleHBsaWNpdGx5IHNlcGFyYXRlLiBDb21wYXJhdG9yIHNldHVwcyBmcm9tIGNpdGVkIHByb3ZpZGVyIHJlcG9ydC4","id":"launch-amazon-1671","ingestRunId":"launch-amazon","modelId":"amazon-nova-premier","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report.","family":"SWE-bench Verified","locator":"Table4","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":42.4,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"terminal-bench-1-0","evidenceKind":"lab_self_report","harnessId":"terminal-bench-1-0:amazon:Tm92YTIgaW50ZXJuYWwgYWdlbnRpYyBzY2FmZm9sZDsgc2NhbGVkIGluZmVyZW5jZSBleHBsaWNpdGx5IHNlcGFyYXRlLiBDb21wYXJhdG9yIHNldHVwcyBmcm9tIGNpdGVkIHByb3ZpZGVyIHJlcG9ydC4","id":"launch-amazon-1672","ingestRunId":"launch-amazon","modelId":"amazon-nova-premier","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report.","family":"Terminal-Bench","locator":"Table4","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":11.3,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"livecodebench-v5-july2024-january2025-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"livecodebench-v5-july2024-january2025-source-release-snapshot-version-not-specified:amazon:Tm92YTIgaW50ZXJuYWwgYWdlbnRpYyBzY2FmZm9sZDsgc2NhbGVkIGluZmVyZW5jZSBleHBsaWNpdGx5IHNlcGFyYXRlLiBDb21wYXJhdG9yIHNldHVwcyBmcm9tIGNpdGVkIHByb3ZpZGVyIHJlcG9ydC4","id":"launch-amazon-1673","ingestRunId":"launch-amazon","modelId":"amazon-nova-premier","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report.","family":"LiveCodeBench","locator":"Table4","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":31.7,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct:amazon:Tm92YTIgaW50ZXJuYWwgYWdlbnRpYyBzY2FmZm9sZDsgc2NhbGVkIGluZmVyZW5jZSBleHBsaWNpdGx5IHNlcGFyYXRlLiBDb21wYXJhdG9yIHNldHVwcyBmcm9tIGNpdGVkIHByb3ZpZGVyIHJlcG9ydC4","id":"launch-amazon-1674","ingestRunId":"launch-amazon","modelId":"claude-sonnet-4-5-20250929","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report.","family":"SWE-bench Verified","locator":"Table4","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":77.2,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"swe-bench-verified-scaled-inference-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"swe-bench-verified-scaled-inference-source-release-snapshot-version-not-specified:amazon:Tm92YTIgaW50ZXJuYWwgYWdlbnRpYyBzY2FmZm9sZDsgc2NhbGVkIGluZmVyZW5jZSBleHBsaWNpdGx5IHNlcGFyYXRlLiBDb21wYXJhdG9yIHNldHVwcyBmcm9tIGNpdGVkIHByb3ZpZGVyIHJlcG9ydC4","id":"launch-amazon-1675","ingestRunId":"launch-amazon","modelId":"claude-sonnet-4-5-20250929","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report.","family":"SWE-bench Verified","locator":"Table4","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":82,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"terminal-bench-1-0","evidenceKind":"lab_self_report","harnessId":"terminal-bench-1-0:amazon:Tm92YTIgaW50ZXJuYWwgYWdlbnRpYyBzY2FmZm9sZDsgc2NhbGVkIGluZmVyZW5jZSBleHBsaWNpdGx5IHNlcGFyYXRlLiBDb21wYXJhdG9yIHNldHVwcyBmcm9tIGNpdGVkIHByb3ZpZGVyIHJlcG9ydC4","id":"launch-amazon-1676","ingestRunId":"launch-amazon","modelId":"claude-sonnet-4-5-20250929","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report.","family":"Terminal-Bench","locator":"Table4","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":51,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"livecodebench-v5-july2024-january2025-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"livecodebench-v5-july2024-january2025-source-release-snapshot-version-not-specified:amazon:Tm92YTIgaW50ZXJuYWwgYWdlbnRpYyBzY2FmZm9sZDsgc2NhbGVkIGluZmVyZW5jZSBleHBsaWNpdGx5IHNlcGFyYXRlLiBDb21wYXJhdG9yIHNldHVwcyBmcm9tIGNpdGVkIHByb3ZpZGVyIHJlcG9ydC4","id":"launch-amazon-1677","ingestRunId":"launch-amazon","modelId":"claude-sonnet-4-5-20250929","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report.","family":"LiveCodeBench","locator":"Table4","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":71.4,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct:amazon:Tm92YTIgaW50ZXJuYWwgYWdlbnRpYyBzY2FmZm9sZDsgc2NhbGVkIGluZmVyZW5jZSBleHBsaWNpdGx5IHNlcGFyYXRlLiBDb21wYXJhdG9yIHNldHVwcyBmcm9tIGNpdGVkIHByb3ZpZGVyIHJlcG9ydC4","id":"launch-amazon-1678","ingestRunId":"launch-amazon","modelId":"gpt-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report.","family":"SWE-bench Verified","locator":"Table4","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":72.8,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"terminal-bench-1-0","evidenceKind":"lab_self_report","harnessId":"terminal-bench-1-0:amazon:Tm92YTIgaW50ZXJuYWwgYWdlbnRpYyBzY2FmZm9sZDsgc2NhbGVkIGluZmVyZW5jZSBleHBsaWNpdGx5IHNlcGFyYXRlLiBDb21wYXJhdG9yIHNldHVwcyBmcm9tIGNpdGVkIHByb3ZpZGVyIHJlcG9ydC4","id":"launch-amazon-1679","ingestRunId":"launch-amazon","modelId":"gpt-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report.","family":"Terminal-Bench","locator":"Table4","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":41.3,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"livecodebench-v5-july2024-january2025-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"livecodebench-v5-july2024-january2025-source-release-snapshot-version-not-specified:amazon:Tm92YTIgaW50ZXJuYWwgYWdlbnRpYyBzY2FmZm9sZDsgc2NhbGVkIGluZmVyZW5jZSBleHBsaWNpdGx5IHNlcGFyYXRlLiBDb21wYXJhdG9yIHNldHVwcyBmcm9tIGNpdGVkIHByb3ZpZGVyIHJlcG9ydC4","id":"launch-amazon-1680","ingestRunId":"launch-amazon","modelId":"gpt-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report.","family":"LiveCodeBench","locator":"Table4","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":84.6,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct:amazon:Tm92YTIgaW50ZXJuYWwgYWdlbnRpYyBzY2FmZm9sZDsgc2NhbGVkIGluZmVyZW5jZSBleHBsaWNpdGx5IHNlcGFyYXRlLiBDb21wYXJhdG9yIHNldHVwcyBmcm9tIGNpdGVkIHByb3ZpZGVyIHJlcG9ydC4","id":"launch-amazon-1681","ingestRunId":"launch-amazon","modelId":"gpt-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report.","family":"SWE-bench Verified","locator":"Table4","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":76.3,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"terminal-bench-1-0","evidenceKind":"lab_self_report","harnessId":"terminal-bench-1-0:amazon:Tm92YTIgaW50ZXJuYWwgYWdlbnRpYyBzY2FmZm9sZDsgc2NhbGVkIGluZmVyZW5jZSBleHBsaWNpdGx5IHNlcGFyYXRlLiBDb21wYXJhdG9yIHNldHVwcyBmcm9tIGNpdGVkIHByb3ZpZGVyIHJlcG9ydC4","id":"launch-amazon-1682","ingestRunId":"launch-amazon","modelId":"gpt-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report.","family":"Terminal-Bench","locator":"Table4","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":53.8,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"livecodebench-v5-july2024-january2025-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"livecodebench-v5-july2024-january2025-source-release-snapshot-version-not-specified:amazon:Tm92YTIgaW50ZXJuYWwgYWdlbnRpYyBzY2FmZm9sZDsgc2NhbGVkIGluZmVyZW5jZSBleHBsaWNpdGx5IHNlcGFyYXRlLiBDb21wYXJhdG9yIHNldHVwcyBmcm9tIGNpdGVkIHByb3ZpZGVyIHJlcG9ydC4","id":"launch-amazon-1683","ingestRunId":"launch-amazon","modelId":"gpt-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report.","family":"LiveCodeBench","locator":"Table4","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":86.8,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct:amazon:Tm92YTIgaW50ZXJuYWwgYWdlbnRpYyBzY2FmZm9sZDsgc2NhbGVkIGluZmVyZW5jZSBleHBsaWNpdGx5IHNlcGFyYXRlLiBDb21wYXJhdG9yIHNldHVwcyBmcm9tIGNpdGVkIHByb3ZpZGVyIHJlcG9ydC4","id":"launch-amazon-1684","ingestRunId":"launch-amazon","modelId":"gemini-2.5-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report.","family":"SWE-bench Verified","locator":"Table4","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":59.6,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"swe-bench-verified-scaled-inference-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"swe-bench-verified-scaled-inference-source-release-snapshot-version-not-specified:amazon:Tm92YTIgaW50ZXJuYWwgYWdlbnRpYyBzY2FmZm9sZDsgc2NhbGVkIGluZmVyZW5jZSBleHBsaWNpdGx5IHNlcGFyYXRlLiBDb21wYXJhdG9yIHNldHVwcyBmcm9tIGNpdGVkIHByb3ZpZGVyIHJlcG9ydC4","id":"launch-amazon-1685","ingestRunId":"launch-amazon","modelId":"gemini-2.5-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report.","family":"SWE-bench Verified","locator":"Table4","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":67.2,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"terminal-bench-1-0","evidenceKind":"lab_self_report","harnessId":"terminal-bench-1-0:amazon:Tm92YTIgaW50ZXJuYWwgYWdlbnRpYyBzY2FmZm9sZDsgc2NhbGVkIGluZmVyZW5jZSBleHBsaWNpdGx5IHNlcGFyYXRlLiBDb21wYXJhdG9yIHNldHVwcyBmcm9tIGNpdGVkIHByb3ZpZGVyIHJlcG9ydC4","id":"launch-amazon-1686","ingestRunId":"launch-amazon","modelId":"gemini-2.5-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report.","family":"Terminal-Bench","locator":"Table4","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":25.3,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"livecodebench-v5-july2024-january2025-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"livecodebench-v5-july2024-january2025-source-release-snapshot-version-not-specified:amazon:Tm92YTIgaW50ZXJuYWwgYWdlbnRpYyBzY2FmZm9sZDsgc2NhbGVkIGluZmVyZW5jZSBleHBsaWNpdGx5IHNlcGFyYXRlLiBDb21wYXJhdG9yIHNldHVwcyBmcm9tIGNpdGVkIHByb3ZpZGVyIHJlcG9ydC4","id":"launch-amazon-1687","ingestRunId":"launch-amazon","modelId":"gemini-2.5-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report.","family":"LiveCodeBench","locator":"Table4","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":80.1,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct:amazon:Tm92YTIgaW50ZXJuYWwgYWdlbnRpYyBzY2FmZm9sZDsgc2NhbGVkIGluZmVyZW5jZSBleHBsaWNpdGx5IHNlcGFyYXRlLiBDb21wYXJhdG9yIHNldHVwcyBmcm9tIGNpdGVkIHByb3ZpZGVyIHJlcG9ydC4","id":"launch-amazon-1688","ingestRunId":"launch-amazon","modelId":"amazon-nova-2-pro-preview","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report.","family":"SWE-bench Verified","locator":"Table4","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":61.5,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"swe-bench-verified-scaled-inference-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"swe-bench-verified-scaled-inference-source-release-snapshot-version-not-specified:amazon:Tm92YTIgaW50ZXJuYWwgYWdlbnRpYyBzY2FmZm9sZDsgc2NhbGVkIGluZmVyZW5jZSBleHBsaWNpdGx5IHNlcGFyYXRlLiBDb21wYXJhdG9yIHNldHVwcyBmcm9tIGNpdGVkIHByb3ZpZGVyIHJlcG9ydC4","id":"launch-amazon-1689","ingestRunId":"launch-amazon","modelId":"amazon-nova-2-pro-preview","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report.","family":"SWE-bench Verified","locator":"Table4","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":70,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"terminal-bench-1-0","evidenceKind":"lab_self_report","harnessId":"terminal-bench-1-0:amazon:Tm92YTIgaW50ZXJuYWwgYWdlbnRpYyBzY2FmZm9sZDsgc2NhbGVkIGluZmVyZW5jZSBleHBsaWNpdGx5IHNlcGFyYXRlLiBDb21wYXJhdG9yIHNldHVwcyBmcm9tIGNpdGVkIHByb3ZpZGVyIHJlcG9ydC4","id":"launch-amazon-1690","ingestRunId":"launch-amazon","modelId":"amazon-nova-2-pro-preview","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report.","family":"Terminal-Bench","locator":"Table4","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":41.3,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"livecodebench-v5-july2024-january2025-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"livecodebench-v5-july2024-january2025-source-release-snapshot-version-not-specified:amazon:Tm92YTIgaW50ZXJuYWwgYWdlbnRpYyBzY2FmZm9sZDsgc2NhbGVkIGluZmVyZW5jZSBleHBsaWNpdGx5IHNlcGFyYXRlLiBDb21wYXJhdG9yIHNldHVwcyBmcm9tIGNpdGVkIHByb3ZpZGVyIHJlcG9ydC4","id":"launch-amazon-1691","ingestRunId":"launch-amazon","modelId":"amazon-nova-2-pro-preview","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report.","family":"LiveCodeBench","locator":"Table4","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"amazon","sourceTitle":"Amazon Nova 2 technical report"},"score":74.6,"scoreUnit":"percent","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf"},{"benchmarkId":"swe-bench-pro-percent-higher","evidenceKind":"lab_self_report","harnessId":"swe-bench-pro-percent-higher:anthropic-fable-5-1-card:QW50aHJvcGljIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IGFkYXB0aXZlIHRoaW5raW5nIGF0IG1heCBlZmZvcnQsIGRlZmF1bHQgc2FtcGxpbmcsIGZpdmUtdHJpYWwgbWVhbiB1bmxlc3Mgc2VjdGlvbiBzcGVjaWZpZXMgb3RoZXJ3aXNlOyBjb250ZXh0IGF0IG1vc3QgMU0uIENvbXBhcmF0b3IgY29uZmlndXJhdGlvbiBmb2xsb3dzIGNpdGVkIHByaW9yIGNhcmQgb3IgYm9hcmQuIEFnZW50IGlkZW50aXR5IGlzIG5vdCBlc3RhYmxpc2hlZCBhcyBtaW5pLXN3ZS1hZ2VudC4","id":"launch-anthropic-fable-5-1-card-0","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"coding","configuration":"Anthropic reported configuration; adaptive thinking at max effort, default sampling, five-trial mean unless section specifies otherwise; context at most 1M. Comparator configuration follows cited prior card or board. Agent identity is not established as mini-swe-agent.","family":"SWE-bench Pro","locator":"Table 8.1.A; section 8.2","metric":"percent","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":81.2,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"swe-bench-pro-percent-higher","evidenceKind":"lab_self_report","harnessId":"swe-bench-pro-percent-higher:anthropic-fable-5-1-card:QW50aHJvcGljIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IGFkYXB0aXZlIHRoaW5raW5nIGF0IG1heCBlZmZvcnQsIGRlZmF1bHQgc2FtcGxpbmcsIGZpdmUtdHJpYWwgbWVhbiB1bmxlc3Mgc2VjdGlvbiBzcGVjaWZpZXMgb3RoZXJ3aXNlOyBjb250ZXh0IGF0IG1vc3QgMU0uIENvbXBhcmF0b3IgY29uZmlndXJhdGlvbiBmb2xsb3dzIGNpdGVkIHByaW9yIGNhcmQgb3IgYm9hcmQuIEFnZW50IGlkZW50aXR5IGlzIG5vdCBlc3RhYmxpc2hlZCBhcyBtaW5pLXN3ZS1hZ2VudC4","id":"launch-anthropic-fable-5-1-card-1","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"coding","configuration":"Anthropic reported configuration; adaptive thinking at max effort, default sampling, five-trial mean unless section specifies otherwise; context at most 1M. Comparator configuration follows cited prior card or board. Agent identity is not established as mini-swe-agent.","family":"SWE-bench Pro","locator":"Table 8.1.A; section 8.2","metric":"percent","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":80,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"deepswe-1-1-percent","evidenceKind":"lab_self_report","harnessId":"deepswe-1-1-percent:anthropic-fable-5-1-card:QW50aHJvcGljIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IGFkYXB0aXZlIHRoaW5raW5nIGF0IG1heCBlZmZvcnQsIGRlZmF1bHQgc2FtcGxpbmcsIGZpdmUtdHJpYWwgbWVhbiB1bmxlc3Mgc2VjdGlvbiBzcGVjaWZpZXMgb3RoZXJ3aXNlOyBjb250ZXh0IGF0IG1vc3QgMU0uIENvbXBhcmF0b3IgY29uZmlndXJhdGlvbiBmb2xsb3dzIGNpdGVkIHByaW9yIGNhcmQgb3IgYm9hcmQuIDExMyB0YXNrczsgb3JpZ2luYWwgaGlkZGVuLXRlc3QgZ3JhZGluZy4","id":"launch-anthropic-fable-5-1-card-10","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"coding","configuration":"Anthropic reported configuration; adaptive thinking at max effort, default sampling, five-trial mean unless section specifies otherwise; context at most 1M. Comparator configuration follows cited prior card or board. 113 tasks; original hidden-test grading.","family":"DeepSWE","locator":"8.3","metric":"percent","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":67.4,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"arc-agi-1-percent","evidenceKind":"lab_self_report","harnessId":"arc-agi-1-percent:anthropic-fable-5-1-card:QVJDIFByaXplIHNlbWktcHJpdmF0ZSB2YWxpZGF0aW9uOyB2ZXJpZmllZCBGYWJsZTUuMSBtYXggZWZmb3J0OyBjb21wYXJhdG9yIGZpZ3VyZXMgYXMgcmVwb3J0ZWQgaW4gc3VtbWFyeS4","id":"launch-anthropic-fable-5-1-card-100","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"hard_reasoning","configuration":"ARC Prize semi-private validation; verified Fable5.1 max effort; comparator figures as reported in summary.","family":"ARC-AGI","locator":"Table8.1.A;8.16","metric":"percent","notes":"","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":96.5,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"arc-agi-2-percent-higher","evidenceKind":"lab_self_report","harnessId":"arc-agi-2-percent-higher:anthropic-fable-5-1-card:QVJDIFByaXplIHNlbWktcHJpdmF0ZSB2YWxpZGF0aW9uOyB2ZXJpZmllZCBGYWJsZTUuMSBtYXggZWZmb3J0OyBjb21wYXJhdG9yIGZpZ3VyZXMgYXMgcmVwb3J0ZWQgaW4gc3VtbWFyeS4","id":"launch-anthropic-fable-5-1-card-101","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"hard_reasoning","configuration":"ARC Prize semi-private validation; verified Fable5.1 max effort; comparator figures as reported in summary.","family":"ARC-AGI","locator":"Table8.1.A;8.16","metric":"percent","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":90,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"arc-agi-2-percent-higher","evidenceKind":"lab_self_report","harnessId":"arc-agi-2-percent-higher:anthropic-fable-5-1-card:QVJDIFByaXplIHNlbWktcHJpdmF0ZSB2YWxpZGF0aW9uOyB2ZXJpZmllZCBGYWJsZTUuMSBtYXggZWZmb3J0OyBjb21wYXJhdG9yIGZpZ3VyZXMgYXMgcmVwb3J0ZWQgaW4gc3VtbWFyeS4","id":"launch-anthropic-fable-5-1-card-102","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"hard_reasoning","configuration":"ARC Prize semi-private validation; verified Fable5.1 max effort; comparator figures as reported in summary.","family":"ARC-AGI","locator":"Table8.1.A;8.16","metric":"percent","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":89.2,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"arc-agi-2-percent-higher","evidenceKind":"lab_self_report","harnessId":"arc-agi-2-percent-higher:anthropic-fable-5-1-card:QVJDIFByaXplIHNlbWktcHJpdmF0ZSB2YWxpZGF0aW9uOyB2ZXJpZmllZCBGYWJsZTUuMSBtYXggZWZmb3J0OyBjb21wYXJhdG9yIGZpZ3VyZXMgYXMgcmVwb3J0ZWQgaW4gc3VtbWFyeS4","id":"launch-anthropic-fable-5-1-card-103","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"hard_reasoning","configuration":"ARC Prize semi-private validation; verified Fable5.1 max effort; comparator figures as reported in summary.","family":"ARC-AGI","locator":"Table8.1.A;8.16","metric":"percent","notes":"","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":90.42,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"arc-agi-2-percent-higher","evidenceKind":"lab_self_report","harnessId":"arc-agi-2-percent-higher:anthropic-fable-5-1-card:QVJDIFByaXplIHNlbWktcHJpdmF0ZSB2YWxpZGF0aW9uOyB2ZXJpZmllZCBGYWJsZTUuMSBtYXggZWZmb3J0OyBjb21wYXJhdG9yIGZpZ3VyZXMgYXMgcmVwb3J0ZWQgaW4gc3VtbWFyeS4","id":"launch-anthropic-fable-5-1-card-104","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"hard_reasoning","configuration":"ARC Prize semi-private validation; verified Fable5.1 max effort; comparator figures as reported in summary.","family":"ARC-AGI","locator":"Table8.1.A;8.16","metric":"percent","notes":"","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":92.5,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"healthbench","evidenceKind":"lab_self_report","harnessId":"healthbench:anthropic-fable-5-1-card:UmF3IHJ1YnJpYyBzY29yZTsgYWRhcHRpdmUgbWF4OyBmaXZlIHRyaWFsczsgbm8gdG9vbHMvY3VzdG9tIHN5c3RlbSBwcm9tcHQ7IE9wdXM0LjhncmFkZXI7IEZhYmxlNS4xIHNhZmV0eSBmYWxsYmFjayB0b09wdXM1Lg","id":"launch-anthropic-fable-5-1-card-105","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"knowledge","configuration":"Raw rubric score; adaptive max; five trials; no tools/custom system prompt; Opus4.8grader; Fable5.1 safety fallback toOpus5.","family":"HealthBench","locator":"8.17","metric":"percent","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":66.7,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"healthbench","evidenceKind":"lab_self_report","harnessId":"healthbench:anthropic-fable-5-1-card:UmF3IHJ1YnJpYyBzY29yZTsgYWRhcHRpdmUgbWF4OyBmaXZlIHRyaWFsczsgbm8gdG9vbHMvY3VzdG9tIHN5c3RlbSBwcm9tcHQ7IE9wdXM0LjhncmFkZXI7IEZhYmxlNS4xIHNhZmV0eSBmYWxsYmFjayB0b09wdXM1Lg","id":"launch-anthropic-fable-5-1-card-106","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"knowledge","configuration":"Raw rubric score; adaptive max; five trials; no tools/custom system prompt; Opus4.8grader; Fable5.1 safety fallback toOpus5.","family":"HealthBench","locator":"8.17","metric":"percent","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":61.2,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"healthbench","evidenceKind":"lab_self_report","harnessId":"healthbench:anthropic-fable-5-1-card:UmF3IHJ1YnJpYyBzY29yZTsgYWRhcHRpdmUgbWF4OyBmaXZlIHRyaWFsczsgbm8gdG9vbHMvY3VzdG9tIHN5c3RlbSBwcm9tcHQ7IE9wdXM0LjhncmFkZXI7IEZhYmxlNS4xIHNhZmV0eSBmYWxsYmFjayB0b09wdXM1Lg","id":"launch-anthropic-fable-5-1-card-107","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"knowledge","configuration":"Raw rubric score; adaptive max; five trials; no tools/custom system prompt; Opus4.8grader; Fable5.1 safety fallback toOpus5.","family":"HealthBench","locator":"8.17","metric":"percent","notes":"","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":67.1,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"healthbench-professional-percent","evidenceKind":"lab_self_report","harnessId":"healthbench-professional-percent:anthropic-fable-5-1-card:UmF3IHJ1YnJpYyBzY29yZTsgYWRhcHRpdmUgbWF4OyBmaXZlIHRyaWFsczsgbm8gdG9vbHMvY3VzdG9tIHN5c3RlbSBwcm9tcHQ7IE9wdXM0LjhncmFkZXI7IEZhYmxlNS4xIHNhZmV0eSBmYWxsYmFjayB0b09wdXM1Lg","id":"launch-anthropic-fable-5-1-card-108","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"knowledge","configuration":"Raw rubric score; adaptive max; five trials; no tools/custom system prompt; Opus4.8grader; Fable5.1 safety fallback toOpus5.","family":"HealthBench Professional","locator":"8.17","metric":"percent","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":74.2,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"healthbench-professional-percent","evidenceKind":"lab_self_report","harnessId":"healthbench-professional-percent:anthropic-fable-5-1-card:UmF3IHJ1YnJpYyBzY29yZTsgYWRhcHRpdmUgbWF4OyBmaXZlIHRyaWFsczsgbm8gdG9vbHMvY3VzdG9tIHN5c3RlbSBwcm9tcHQ7IE9wdXM0LjhncmFkZXI7IEZhYmxlNS4xIHNhZmV0eSBmYWxsYmFjayB0b09wdXM1Lg","id":"launch-anthropic-fable-5-1-card-109","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"knowledge","configuration":"Raw rubric score; adaptive max; five trials; no tools/custom system prompt; Opus4.8grader; Fable5.1 safety fallback toOpus5.","family":"HealthBench Professional","locator":"8.17","metric":"percent","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":68.9,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"frontiercode-1-1-extended","evidenceKind":"lab_self_report","harnessId":"frontiercode-1-1-extended:anthropic-fable-5-1-card:Q29nbml0aW9uIGFnZW50aWMgY29kaW5nOyBjb21wb3NpdGUgZnVuY3Rpb25hbCBhbmQgY29kZS1xdWFsaXR5IHNjb3JlLiBGYWJsZTUuMSBtZWRpdW0gZWZmb3J0OyBGYWJsZTUgeGhpZ2gu","id":"launch-anthropic-fable-5-1-card-11","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"coding","configuration":"Cognition agentic coding; composite functional and code-quality score. Fable5.1 medium effort; Fable5 xhigh.","family":"FrontierCode","locator":"8.4","metric":"percent","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":63.6,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"healthbench-professional-percent","evidenceKind":"lab_self_report","harnessId":"healthbench-professional-percent:anthropic-fable-5-1-card:UmF3IHJ1YnJpYyBzY29yZTsgYWRhcHRpdmUgbWF4OyBmaXZlIHRyaWFsczsgbm8gdG9vbHMvY3VzdG9tIHN5c3RlbSBwcm9tcHQ7IE9wdXM0LjhncmFkZXI7IEZhYmxlNS4xIHNhZmV0eSBmYWxsYmFjayB0b09wdXM1Lg","id":"launch-anthropic-fable-5-1-card-110","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"knowledge","configuration":"Raw rubric score; adaptive max; five trials; no tools/custom system prompt; Opus4.8grader; Fable5.1 safety fallback toOpus5.","family":"HealthBench Professional","locator":"8.17","metric":"percent","notes":"","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":73.4,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"healthbench","evidenceKind":"lab_self_report","harnessId":"healthbench:anthropic-fable-5-1-card:TGVuZ3RoLWFkanVzdGVkIHNjb3JlIHVzaW5nIEdQVDUuNWNhcmQgbWV0aG9kOyBvdGhlcndpc2UgcmF3IGV2YWx1YXRpb24gY29uZmlndXJhdGlvbi4","id":"launch-anthropic-fable-5-1-card-111","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"knowledge","configuration":"Length-adjusted score using GPT5.5card method; otherwise raw evaluation configuration.","family":"HealthBench","locator":"8.17.1","metric":"percent","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":60,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"healthbench-professional-percent","evidenceKind":"lab_self_report","harnessId":"healthbench-professional-percent:anthropic-fable-5-1-card:TGVuZ3RoLWFkanVzdGVkIHNjb3JlOyBIZWFsdGhCZW5jaCBQcm9mZXNzaW9uYWwgcGFwZXIgbWV0aG9kOyBubyB0b29sczsgT3B1czQuOGdyYWRlcjsgZml2ZSB0cmlhbHMu","id":"launch-anthropic-fable-5-1-card-112","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"knowledge","configuration":"Length-adjusted score; HealthBench Professional paper method; no tools; Opus4.8grader; five trials.","family":"HealthBench Professional","locator":"Table8.1.A;8.17.2","metric":"percent","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":62.1,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"healthbench-professional-percent","evidenceKind":"lab_self_report","harnessId":"healthbench-professional-percent:anthropic-fable-5-1-card:TGVuZ3RoLWFkanVzdGVkIHNjb3JlOyBIZWFsdGhCZW5jaCBQcm9mZXNzaW9uYWwgcGFwZXIgbWV0aG9kOyBubyB0b29sczsgT3B1czQuOGdyYWRlcjsgZml2ZSB0cmlhbHMu","id":"launch-anthropic-fable-5-1-card-113","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"knowledge","configuration":"Length-adjusted score; HealthBench Professional paper method; no tools; Opus4.8grader; five trials.","family":"HealthBench Professional","locator":"Table8.1.A;8.17.2","metric":"percent","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":63.3,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"healthbench-professional-percent","evidenceKind":"lab_self_report","harnessId":"healthbench-professional-percent:anthropic-fable-5-1-card:TGVuZ3RoLWFkanVzdGVkIHNjb3JlOyBIZWFsdGhCZW5jaCBQcm9mZXNzaW9uYWwgcGFwZXIgbWV0aG9kOyBubyB0b29sczsgT3B1czQuOGdyYWRlcjsgZml2ZSB0cmlhbHMu","id":"launch-anthropic-fable-5-1-card-114","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"knowledge","configuration":"Length-adjusted score; HealthBench Professional paper method; no tools; Opus4.8grader; five trials.","family":"HealthBench Professional","locator":"Table8.1.A;8.17.2","metric":"percent","notes":"","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":59.8,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"gmmlu","evidenceKind":"lab_self_report","harnessId":"gmmlu:anthropic-fable-5-1-card:TWVhbiBhY2N1cmFjeTQybGFuZ3VhZ2VzOyBhZGFwdGl2ZSBtYXg7IG9uZSB0cmlhbDsgbm8gdG9vbHMvY3VzdG9tIHN5c3RlbSBwcm9tcHRzLg","id":"launch-anthropic-fable-5-1-card-115","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"knowledge","configuration":"Mean accuracy42languages; adaptive max; one trial; no tools/custom system prompts.","family":"GMMLU","locator":"8.18.1","metric":"percent","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":94,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"gmmlu","evidenceKind":"lab_self_report","harnessId":"gmmlu:anthropic-fable-5-1-card:TWVhbiBhY2N1cmFjeTQybGFuZ3VhZ2VzOyBhZGFwdGl2ZSBtYXg7IG9uZSB0cmlhbDsgbm8gdG9vbHMvY3VzdG9tIHN5c3RlbSBwcm9tcHRzLg","id":"launch-anthropic-fable-5-1-card-116","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"knowledge","configuration":"Mean accuracy42languages; adaptive max; one trial; no tools/custom system prompts.","family":"GMMLU","locator":"8.18.1","metric":"percent","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":93.6,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"gmmlu","evidenceKind":"lab_self_report","harnessId":"gmmlu:anthropic-fable-5-1-card:TWVhbiBhY2N1cmFjeTQybGFuZ3VhZ2VzOyBhZGFwdGl2ZSBtYXg7IG9uZSB0cmlhbDsgbm8gdG9vbHMvY3VzdG9tIHN5c3RlbSBwcm9tcHRzLg","id":"launch-anthropic-fable-5-1-card-117","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"knowledge","configuration":"Mean accuracy42languages; adaptive max; one trial; no tools/custom system prompts.","family":"GMMLU","locator":"8.18.1","metric":"percent","notes":"","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":92.5,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"milu","evidenceKind":"lab_self_report","harnessId":"milu:anthropic-fable-5-1-card:TWVhbiBhY2N1cmFjeTExbGFuZ3VhZ2VzOyBhZGFwdGl2ZSBtYXg7IGZpdmUgdHJpYWxzOyBubyB0b29scy9jdXN0b20gc3lzdGVtIHByb21wdHMu","id":"launch-anthropic-fable-5-1-card-118","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"knowledge","configuration":"Mean accuracy11languages; adaptive max; five trials; no tools/custom system prompts.","family":"MILU","locator":"8.18.2","metric":"percent","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":93,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"milu","evidenceKind":"lab_self_report","harnessId":"milu:anthropic-fable-5-1-card:TWVhbiBhY2N1cmFjeTExbGFuZ3VhZ2VzOyBhZGFwdGl2ZSBtYXg7IGZpdmUgdHJpYWxzOyBubyB0b29scy9jdXN0b20gc3lzdGVtIHByb21wdHMu","id":"launch-anthropic-fable-5-1-card-119","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"knowledge","configuration":"Mean accuracy11languages; adaptive max; five trials; no tools/custom system prompts.","family":"MILU","locator":"8.18.2","metric":"percent","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":92.9,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"frontiercode-1-1-extended","evidenceKind":"lab_self_report","harnessId":"frontiercode-1-1-extended:anthropic-fable-5-1-card:Q29nbml0aW9uIGFnZW50aWMgY29kaW5nOyBjb21wb3NpdGUgZnVuY3Rpb25hbCBhbmQgY29kZS1xdWFsaXR5IHNjb3JlLiBGYWJsZTUuMSBtZWRpdW0gZWZmb3J0OyBGYWJsZTUgeGhpZ2gu","id":"launch-anthropic-fable-5-1-card-12","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"coding","configuration":"Cognition agentic coding; composite functional and code-quality score. Fable5.1 medium effort; Fable5 xhigh.","family":"FrontierCode","locator":"8.4","metric":"percent","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":64.9,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"milu","evidenceKind":"lab_self_report","harnessId":"milu:anthropic-fable-5-1-card:TWVhbiBhY2N1cmFjeTExbGFuZ3VhZ2VzOyBhZGFwdGl2ZSBtYXg7IGZpdmUgdHJpYWxzOyBubyB0b29scy9jdXN0b20gc3lzdGVtIHByb21wdHMu","id":"launch-anthropic-fable-5-1-card-120","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"knowledge","configuration":"Mean accuracy11languages; adaptive max; five trials; no tools/custom system prompts.","family":"MILU","locator":"8.18.2","metric":"percent","notes":"","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":92.1,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"biomysterybench-human-solvable","evidenceKind":"lab_self_report","harnessId":"biomysterybench-human-solvable:anthropic-fable-5-1-card:QW50aHJvcGljIGxpZmUtc2NpZW5jZXMgZXZhbHVhdGlvbjsgYmFzaC9lZGl0b3IvcGFja2FnZXMgZXhjZXB0IHByb3RlaW4gZGVzaWduIG5vIHRvb2xzOyBwcm90b2NvbHMgdXNlIGJhc2gvZWRpdG9yL3NlYXJjaDsgVW5kZXJzdGFuZGluZyByZXZpc2VkIG5ldHdvcmsgcmVzdHJpY3Rpb24u","id":"launch-anthropic-fable-5-1-card-121","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"knowledge","configuration":"Anthropic life-sciences evaluation; bash/editor/packages except protein design no tools; protocols use bash/editor/search; Understanding revised network restriction.","family":"BioMysteryBench","locator":"8.19","metric":"percent","notes":"Internal evaluation; not necessarily publicly released; compare only same task/grader revision. Percent rank correlation is not accuracy.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":91.4,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"biomysterybench-human-solvable","evidenceKind":"lab_self_report","harnessId":"biomysterybench-human-solvable:anthropic-fable-5-1-card:QW50aHJvcGljIGxpZmUtc2NpZW5jZXMgZXZhbHVhdGlvbjsgYmFzaC9lZGl0b3IvcGFja2FnZXMgZXhjZXB0IHByb3RlaW4gZGVzaWduIG5vIHRvb2xzOyBwcm90b2NvbHMgdXNlIGJhc2gvZWRpdG9yL3NlYXJjaDsgVW5kZXJzdGFuZGluZyByZXZpc2VkIG5ldHdvcmsgcmVzdHJpY3Rpb24u","id":"launch-anthropic-fable-5-1-card-122","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"knowledge","configuration":"Anthropic life-sciences evaluation; bash/editor/packages except protein design no tools; protocols use bash/editor/search; Understanding revised network restriction.","family":"BioMysteryBench","locator":"8.19","metric":"percent","notes":"Internal evaluation; not necessarily publicly released; compare only same task/grader revision. Percent rank correlation is not accuracy.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":86.1,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"biomysterybench-human-difficult","evidenceKind":"lab_self_report","harnessId":"biomysterybench-human-difficult:anthropic-fable-5-1-card:QW50aHJvcGljIGxpZmUtc2NpZW5jZXMgZXZhbHVhdGlvbjsgYmFzaC9lZGl0b3IvcGFja2FnZXMgZXhjZXB0IHByb3RlaW4gZGVzaWduIG5vIHRvb2xzOyBwcm90b2NvbHMgdXNlIGJhc2gvZWRpdG9yL3NlYXJjaDsgVW5kZXJzdGFuZGluZyByZXZpc2VkIG5ldHdvcmsgcmVzdHJpY3Rpb24u","id":"launch-anthropic-fable-5-1-card-123","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"knowledge","configuration":"Anthropic life-sciences evaluation; bash/editor/packages except protein design no tools; protocols use bash/editor/search; Understanding revised network restriction.","family":"BioMysteryBench","locator":"8.19","metric":"percent","notes":"Internal evaluation; not necessarily publicly released; compare only same task/grader revision. Percent rank correlation is not accuracy.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":51.8,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"biomysterybench-human-difficult","evidenceKind":"lab_self_report","harnessId":"biomysterybench-human-difficult:anthropic-fable-5-1-card:QW50aHJvcGljIGxpZmUtc2NpZW5jZXMgZXZhbHVhdGlvbjsgYmFzaC9lZGl0b3IvcGFja2FnZXMgZXhjZXB0IHByb3RlaW4gZGVzaWduIG5vIHRvb2xzOyBwcm90b2NvbHMgdXNlIGJhc2gvZWRpdG9yL3NlYXJjaDsgVW5kZXJzdGFuZGluZyByZXZpc2VkIG5ldHdvcmsgcmVzdHJpY3Rpb24u","id":"launch-anthropic-fable-5-1-card-124","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"knowledge","configuration":"Anthropic life-sciences evaluation; bash/editor/packages except protein design no tools; protocols use bash/editor/search; Understanding revised network restriction.","family":"BioMysteryBench","locator":"8.19","metric":"percent","notes":"Internal evaluation; not necessarily publicly released; compare only same task/grader revision. Percent rank correlation is not accuracy.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":28.8,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"spatialbench-verified","evidenceKind":"lab_self_report","harnessId":"spatialbench-verified:anthropic-fable-5-1-card:QW50aHJvcGljIGxpZmUtc2NpZW5jZXMgZXZhbHVhdGlvbjsgYmFzaC9lZGl0b3IvcGFja2FnZXMgZXhjZXB0IHByb3RlaW4gZGVzaWduIG5vIHRvb2xzOyBwcm90b2NvbHMgdXNlIGJhc2gvZWRpdG9yL3NlYXJjaDsgVW5kZXJzdGFuZGluZyByZXZpc2VkIG5ldHdvcmsgcmVzdHJpY3Rpb24u","id":"launch-anthropic-fable-5-1-card-125","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"knowledge","configuration":"Anthropic life-sciences evaluation; bash/editor/packages except protein design no tools; protocols use bash/editor/search; Understanding revised network restriction.","family":"SpatialBench","locator":"8.19","metric":"percent","notes":"Internal evaluation; not necessarily publicly released; compare only same task/grader revision. Percent rank correlation is not accuracy.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":72.5,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"singlecellbench","evidenceKind":"lab_self_report","harnessId":"singlecellbench:anthropic-fable-5-1-card:QW50aHJvcGljIGxpZmUtc2NpZW5jZXMgZXZhbHVhdGlvbjsgYmFzaC9lZGl0b3IvcGFja2FnZXMgZXhjZXB0IHByb3RlaW4gZGVzaWduIG5vIHRvb2xzOyBwcm90b2NvbHMgdXNlIGJhc2gvZWRpdG9yL3NlYXJjaDsgVW5kZXJzdGFuZGluZyByZXZpc2VkIG5ldHdvcmsgcmVzdHJpY3Rpb24u","id":"launch-anthropic-fable-5-1-card-126","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"knowledge","configuration":"Anthropic life-sciences evaluation; bash/editor/packages except protein design no tools; protocols use bash/editor/search; Understanding revised network restriction.","family":"SingleCellBench","locator":"8.19","metric":"percent","notes":"Internal evaluation; not necessarily publicly released; compare only same task/grader revision. Percent rank correlation is not accuracy.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":60.6,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"proteingym-hard","evidenceKind":"lab_self_report","harnessId":"proteingym-hard:anthropic-fable-5-1-card:QW50aHJvcGljIGxpZmUtc2NpZW5jZXMgZXZhbHVhdGlvbjsgYmFzaC9lZGl0b3IvcGFja2FnZXMgZXhjZXB0IHByb3RlaW4gZGVzaWduIG5vIHRvb2xzOyBwcm90b2NvbHMgdXNlIGJhc2gvZWRpdG9yL3NlYXJjaDsgVW5kZXJzdGFuZGluZyByZXZpc2VkIG5ldHdvcmsgcmVzdHJpY3Rpb24u","id":"launch-anthropic-fable-5-1-card-127","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"knowledge","configuration":"Anthropic life-sciences evaluation; bash/editor/packages except protein design no tools; protocols use bash/editor/search; Understanding revised network restriction.","family":"ProteinGym","locator":"8.19","metric":"percent rank correlation","notes":"Internal evaluation; not necessarily publicly released; compare only same task/grader revision. Percent rank correlation is not accuracy.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":47.7,"scoreUnit":"index","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"proteingym-hard","evidenceKind":"lab_self_report","harnessId":"proteingym-hard:anthropic-fable-5-1-card:QW50aHJvcGljIGxpZmUtc2NpZW5jZXMgZXZhbHVhdGlvbjsgYmFzaC9lZGl0b3IvcGFja2FnZXMgZXhjZXB0IHByb3RlaW4gZGVzaWduIG5vIHRvb2xzOyBwcm90b2NvbHMgdXNlIGJhc2gvZWRpdG9yL3NlYXJjaDsgVW5kZXJzdGFuZGluZyByZXZpc2VkIG5ldHdvcmsgcmVzdHJpY3Rpb24u","id":"launch-anthropic-fable-5-1-card-128","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"knowledge","configuration":"Anthropic life-sciences evaluation; bash/editor/packages except protein design no tools; protocols use bash/editor/search; Understanding revised network restriction.","family":"ProteinGym","locator":"8.19","metric":"percent rank correlation","notes":"Internal evaluation; not necessarily publicly released; compare only same task/grader revision. Percent rank correlation is not accuracy.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":35.5,"scoreUnit":"index","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"protein-design-sequence-generation-revised-grader","evidenceKind":"lab_self_report","harnessId":"protein-design-sequence-generation-revised-grader:anthropic-fable-5-1-card:QW50aHJvcGljIGxpZmUtc2NpZW5jZXMgZXZhbHVhdGlvbjsgYmFzaC9lZGl0b3IvcGFja2FnZXMgZXhjZXB0IHByb3RlaW4gZGVzaWduIG5vIHRvb2xzOyBwcm90b2NvbHMgdXNlIGJhc2gvZWRpdG9yL3NlYXJjaDsgVW5kZXJzdGFuZGluZyByZXZpc2VkIG5ldHdvcmsgcmVzdHJpY3Rpb24u","id":"launch-anthropic-fable-5-1-card-129","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"knowledge","configuration":"Anthropic life-sciences evaluation; bash/editor/packages except protein design no tools; protocols use bash/editor/search; Understanding revised network restriction.","family":"Protein Design","locator":"8.19","metric":"percent","notes":"Internal evaluation; not necessarily publicly released; compare only same task/grader revision. Percent rank correlation is not accuracy.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":42.4,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"frontiercode-1-1-main","evidenceKind":"lab_self_report","harnessId":"frontiercode-1-1-main:anthropic-fable-5-1-card:Q29nbml0aW9uIGFnZW50aWMgY29kaW5nOyBjb21wb3NpdGUgZnVuY3Rpb25hbCBhbmQgY29kZS1xdWFsaXR5IHNjb3JlLiBGYWJsZTUuMSBtZWRpdW0gZWZmb3J0OyBGYWJsZTUgeGhpZ2gu","id":"launch-anthropic-fable-5-1-card-13","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"coding","configuration":"Cognition agentic coding; composite functional and code-quality score. Fable5.1 medium effort; Fable5 xhigh.","family":"FrontierCode","locator":"8.4","metric":"percent","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":50.9,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"protein-design-library-ranking","evidenceKind":"lab_self_report","harnessId":"protein-design-library-ranking:anthropic-fable-5-1-card:QW50aHJvcGljIGxpZmUtc2NpZW5jZXMgZXZhbHVhdGlvbjsgYmFzaC9lZGl0b3IvcGFja2FnZXMgZXhjZXB0IHByb3RlaW4gZGVzaWduIG5vIHRvb2xzOyBwcm90b2NvbHMgdXNlIGJhc2gvZWRpdG9yL3NlYXJjaDsgVW5kZXJzdGFuZGluZyByZXZpc2VkIG5ldHdvcmsgcmVzdHJpY3Rpb24u","id":"launch-anthropic-fable-5-1-card-130","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"knowledge","configuration":"Anthropic life-sciences evaluation; bash/editor/packages except protein design no tools; protocols use bash/editor/search; Understanding revised network restriction.","family":"Protein Design","locator":"8.19","metric":"percent","notes":"Internal evaluation; not necessarily publicly released; compare only same task/grader revision. Percent rank correlation is not accuracy.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":48,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"organic-chemistry-2-revised","evidenceKind":"lab_self_report","harnessId":"organic-chemistry-2-revised:anthropic-fable-5-1-card:QW50aHJvcGljIGxpZmUtc2NpZW5jZXMgZXZhbHVhdGlvbjsgYmFzaC9lZGl0b3IvcGFja2FnZXMgZXhjZXB0IHByb3RlaW4gZGVzaWduIG5vIHRvb2xzOyBwcm90b2NvbHMgdXNlIGJhc2gvZWRpdG9yL3NlYXJjaDsgVW5kZXJzdGFuZGluZyByZXZpc2VkIG5ldHdvcmsgcmVzdHJpY3Rpb24u","id":"launch-anthropic-fable-5-1-card-131","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"knowledge","configuration":"Anthropic life-sciences evaluation; bash/editor/packages except protein design no tools; protocols use bash/editor/search; Understanding revised network restriction.","family":"Organic Chemistry","locator":"8.19","metric":"percent","notes":"Internal evaluation; not necessarily publicly released; compare only same task/grader revision. Percent rank correlation is not accuracy.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":65.7,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"organic-chemistry-2-revised","evidenceKind":"lab_self_report","harnessId":"organic-chemistry-2-revised:anthropic-fable-5-1-card:QW50aHJvcGljIGxpZmUtc2NpZW5jZXMgZXZhbHVhdGlvbjsgYmFzaC9lZGl0b3IvcGFja2FnZXMgZXhjZXB0IHByb3RlaW4gZGVzaWduIG5vIHRvb2xzOyBwcm90b2NvbHMgdXNlIGJhc2gvZWRpdG9yL3NlYXJjaDsgVW5kZXJzdGFuZGluZyByZXZpc2VkIG5ldHdvcmsgcmVzdHJpY3Rpb24u","id":"launch-anthropic-fable-5-1-card-132","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"knowledge","configuration":"Anthropic life-sciences evaluation; bash/editor/packages except protein design no tools; protocols use bash/editor/search; Understanding revised network restriction.","family":"Organic Chemistry","locator":"8.19","metric":"percent","notes":"Internal evaluation; not necessarily publicly released; compare only same task/grader revision. Percent rank correlation is not accuracy.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":43.2,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"protocols-troubleshooting","evidenceKind":"lab_self_report","harnessId":"protocols-troubleshooting:anthropic-fable-5-1-card:QW50aHJvcGljIGxpZmUtc2NpZW5jZXMgZXZhbHVhdGlvbjsgYmFzaC9lZGl0b3IvcGFja2FnZXMgZXhjZXB0IHByb3RlaW4gZGVzaWduIG5vIHRvb2xzOyBwcm90b2NvbHMgdXNlIGJhc2gvZWRpdG9yL3NlYXJjaDsgVW5kZXJzdGFuZGluZyByZXZpc2VkIG5ldHdvcmsgcmVzdHJpY3Rpb24u","id":"launch-anthropic-fable-5-1-card-133","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"knowledge","configuration":"Anthropic life-sciences evaluation; bash/editor/packages except protein design no tools; protocols use bash/editor/search; Understanding revised network restriction.","family":"Protocols","locator":"8.19","metric":"percent","notes":"Internal evaluation; not necessarily publicly released; compare only same task/grader revision. Percent rank correlation is not accuracy.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":61.1,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"protocols-troubleshooting","evidenceKind":"lab_self_report","harnessId":"protocols-troubleshooting:anthropic-fable-5-1-card:QW50aHJvcGljIGxpZmUtc2NpZW5jZXMgZXZhbHVhdGlvbjsgYmFzaC9lZGl0b3IvcGFja2FnZXMgZXhjZXB0IHByb3RlaW4gZGVzaWduIG5vIHRvb2xzOyBwcm90b2NvbHMgdXNlIGJhc2gvZWRpdG9yL3NlYXJjaDsgVW5kZXJzdGFuZGluZyByZXZpc2VkIG5ldHdvcmsgcmVzdHJpY3Rpb24u","id":"launch-anthropic-fable-5-1-card-134","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"knowledge","configuration":"Anthropic life-sciences evaluation; bash/editor/packages except protein design no tools; protocols use bash/editor/search; Understanding revised network restriction.","family":"Protocols","locator":"8.19","metric":"percent","notes":"Internal evaluation; not necessarily publicly released; compare only same task/grader revision. Percent rank correlation is not accuracy.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":56.4,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"protocols-understanding-network-restricted","evidenceKind":"lab_self_report","harnessId":"protocols-understanding-network-restricted:anthropic-fable-5-1-card:QW50aHJvcGljIGxpZmUtc2NpZW5jZXMgZXZhbHVhdGlvbjsgYmFzaC9lZGl0b3IvcGFja2FnZXMgZXhjZXB0IHByb3RlaW4gZGVzaWduIG5vIHRvb2xzOyBwcm90b2NvbHMgdXNlIGJhc2gvZWRpdG9yL3NlYXJjaDsgVW5kZXJzdGFuZGluZyByZXZpc2VkIG5ldHdvcmsgcmVzdHJpY3Rpb24u","id":"launch-anthropic-fable-5-1-card-135","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"knowledge","configuration":"Anthropic life-sciences evaluation; bash/editor/packages except protein design no tools; protocols use bash/editor/search; Understanding revised network restriction.","family":"Protocols","locator":"8.19","metric":"percent","notes":"Internal evaluation; not necessarily publicly released; compare only same task/grader revision. Percent rank correlation is not accuracy.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":80,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"protocols-understanding-network-restricted","evidenceKind":"lab_self_report","harnessId":"protocols-understanding-network-restricted:anthropic-fable-5-1-card:QW50aHJvcGljIGxpZmUtc2NpZW5jZXMgZXZhbHVhdGlvbjsgYmFzaC9lZGl0b3IvcGFja2FnZXMgZXhjZXB0IHByb3RlaW4gZGVzaWduIG5vIHRvb2xzOyBwcm90b2NvbHMgdXNlIGJhc2gvZWRpdG9yL3NlYXJjaDsgVW5kZXJzdGFuZGluZyByZXZpc2VkIG5ldHdvcmsgcmVzdHJpY3Rpb24u","id":"launch-anthropic-fable-5-1-card-136","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"knowledge","configuration":"Anthropic life-sciences evaluation; bash/editor/packages except protein design no tools; protocols use bash/editor/search; Understanding revised network restriction.","family":"Protocols","locator":"8.19","metric":"percent","notes":"Internal evaluation; not necessarily publicly released; compare only same task/grader revision. Percent rank correlation is not accuracy.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":63.9,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"frontiercode-1-1-main","evidenceKind":"lab_self_report","harnessId":"frontiercode-1-1-main:anthropic-fable-5-1-card:Q29nbml0aW9uIGFnZW50aWMgY29kaW5nOyBjb21wb3NpdGUgZnVuY3Rpb25hbCBhbmQgY29kZS1xdWFsaXR5IHNjb3JlLiBGYWJsZTUuMSBtZWRpdW0gZWZmb3J0OyBGYWJsZTUgeGhpZ2gu","id":"launch-anthropic-fable-5-1-card-14","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"coding","configuration":"Cognition agentic coding; composite functional and code-quality score. Fable5.1 medium effort; Fable5 xhigh.","family":"FrontierCode","locator":"8.4","metric":"percent","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":53.5,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"frontierswe-2","evidenceKind":"lab_self_report","harnessId":"frontierswe-2:anthropic-fable-5-1-card:UHJveGltYWwgYWdlbnQgaGFybmVzczsgbWF4IGVmZm9ydDsgMzQgdGFza3MsIGZpdmUgdHJpYWxzL3Rhc2s7IG1lYW4gc2NvcmUgb24gMC4uMSBzY2FsZS4","id":"launch-anthropic-fable-5-1-card-15","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"coding","configuration":"Proximal agent harness; max effort; 34 tasks, five trials/task; mean score on 0..1 scale.","family":"FrontierSWE","locator":"8.5","metric":"fraction","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":0.57,"scoreUnit":"index","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"frontierswe-2","evidenceKind":"lab_self_report","harnessId":"frontierswe-2:anthropic-fable-5-1-card:UHJveGltYWwgYWdlbnQgaGFybmVzczsgbWF4IGVmZm9ydDsgMzQgdGFza3MsIGZpdmUgdHJpYWxzL3Rhc2s7IG1lYW4gc2NvcmUgb24gMC4uMSBzY2FsZS4","id":"launch-anthropic-fable-5-1-card-16","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"coding","configuration":"Proximal agent harness; max effort; 34 tasks, five trials/task; mean score on 0..1 scale.","family":"FrontierSWE","locator":"8.5","metric":"fraction","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":0.48,"scoreUnit":"index","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"frontierswe-2","evidenceKind":"lab_self_report","harnessId":"frontierswe-2:anthropic-fable-5-1-card:UHJveGltYWwgYWdlbnQgaGFybmVzczsgbWF4IGVmZm9ydDsgMzQgdGFza3MsIGZpdmUgdHJpYWxzL3Rhc2s7IG1lYW4gc2NvcmUgb24gMC4uMSBzY2FsZS4","id":"launch-anthropic-fable-5-1-card-17","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"coding","configuration":"Proximal agent harness; max effort; 34 tasks, five trials/task; mean score on 0..1 scale.","family":"FrontierSWE","locator":"8.5","metric":"fraction","notes":"","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":0.52,"scoreUnit":"index","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"frontierswe-2","evidenceKind":"lab_self_report","harnessId":"frontierswe-2:anthropic-fable-5-1-card:UHJveGltYWwgYWdlbnQgaGFybmVzczsgbWF4IGVmZm9ydDsgMzQgdGFza3MsIGZpdmUgdHJpYWxzL3Rhc2s7IG1lYW4gc2NvcmUgb24gMC4uMSBzY2FsZS4","id":"launch-anthropic-fable-5-1-card-18","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"coding","configuration":"Proximal agent harness; max effort; 34 tasks, five trials/task; mean score on 0..1 scale.","family":"FrontierSWE","locator":"8.5","metric":"fraction","notes":"","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":0.32,"scoreUnit":"index","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"terminal-bench-4-0-percent","evidenceKind":"lab_self_report","harnessId":"terminal-bench-4-0-percent:anthropic-fable-5-1-card:Q2xhdWRlIENvZGUgLS1iYXJlIG1heCBlZmZvcnQsIDE1IHRyaWFscy90YXNrIG92ZXIgNjYgdGFza3M7IEFudGhyb3BpYyBpbnRlcm5hbCByZXJ1bnMu","id":"launch-anthropic-fable-5-1-card-19","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"agentic","configuration":"Claude Code --bare max effort, 15 trials/task over 66 tasks; Anthropic internal reruns.","family":"Terminal-Bench","locator":"8.6","metric":"percent","notes":"Uses more precise section values rather than rounded headline table. Distinct from2.1; timeouts/resources changed. Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":55.8,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"swe-bench-pro-percent-higher","evidenceKind":"lab_self_report","harnessId":"swe-bench-pro-percent-higher:anthropic-fable-5-1-card:QW50aHJvcGljIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IGFkYXB0aXZlIHRoaW5raW5nIGF0IG1heCBlZmZvcnQsIGRlZmF1bHQgc2FtcGxpbmcsIGZpdmUtdHJpYWwgbWVhbiB1bmxlc3Mgc2VjdGlvbiBzcGVjaWZpZXMgb3RoZXJ3aXNlOyBjb250ZXh0IGF0IG1vc3QgMU0uIENvbXBhcmF0b3IgY29uZmlndXJhdGlvbiBmb2xsb3dzIGNpdGVkIHByaW9yIGNhcmQgb3IgYm9hcmQuIEFnZW50IGlkZW50aXR5IGlzIG5vdCBlc3RhYmxpc2hlZCBhcyBtaW5pLXN3ZS1hZ2VudC4","id":"launch-anthropic-fable-5-1-card-2","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"coding","configuration":"Anthropic reported configuration; adaptive thinking at max effort, default sampling, five-trial mean unless section specifies otherwise; context at most 1M. Comparator configuration follows cited prior card or board. Agent identity is not established as mini-swe-agent.","family":"SWE-bench Pro","locator":"Table 8.1.A; section 8.2","metric":"percent","notes":"","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":79.2,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"terminal-bench-4-0-percent","evidenceKind":"lab_self_report","harnessId":"terminal-bench-4-0-percent:anthropic-fable-5-1-card:Q2xhdWRlIENvZGUgLS1iYXJlIG1heCBlZmZvcnQsIDE1IHRyaWFscy90YXNrIG92ZXIgNjYgdGFza3M7IEFudGhyb3BpYyBpbnRlcm5hbCByZXJ1bnMu","id":"launch-anthropic-fable-5-1-card-20","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"agentic","configuration":"Claude Code --bare max effort, 15 trials/task over 66 tasks; Anthropic internal reruns.","family":"Terminal-Bench","locator":"8.6","metric":"percent","notes":"Uses more precise section values rather than rounded headline table. Distinct from2.1; timeouts/resources changed. Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":42,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"terminal-bench-4-0-percent","evidenceKind":"lab_self_report","harnessId":"terminal-bench-4-0-percent:anthropic-fable-5-1-card:Q2xhdWRlIENvZGUgLS1iYXJlIG1heCBlZmZvcnQsIDE1IHRyaWFscy90YXNrIG92ZXIgNjYgdGFza3M7IEFudGhyb3BpYyBpbnRlcm5hbCByZXJ1bnMu","id":"launch-anthropic-fable-5-1-card-21","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"agentic","configuration":"Claude Code --bare max effort, 15 trials/task over 66 tasks; Anthropic internal reruns.","family":"Terminal-Bench","locator":"8.6","metric":"percent","notes":"Uses more precise section values rather than rounded headline table. Distinct from2.1; timeouts/resources changed.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":52.3,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"terminal-bench-4-0-percent","evidenceKind":"lab_self_report","harnessId":"terminal-bench-4-0-percent:anthropic-fable-5-1-card:Q2xhdWRlIENvZGUgLS1iYXJlIG1heCBlZmZvcnQsIDE1IHRyaWFscy90YXNrIG92ZXI2NnRhc2tzIGZvciBDbGF1ZGU7IEdQVCBDb2RleCBDTEkgbWF4IGZyb20gcHVibGljIGJvYXJkLg","id":"launch-anthropic-fable-5-1-card-22","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"agentic","comparabilityReason":"System card section 8.6 cites the public Codex CLI result for Sol, while the Claude rows are internal Claude Code --bare reruns. Different agent harnesses and runs cannot form a matched comparison.","comparable":false,"configuration":"Claude Code --bare max effort, 15 trials/task over66tasks for Claude; GPT Codex CLI max from public board.","family":"Terminal-Bench","locator":"8.6","metric":"percent","notes":"Uses more precise section values rather than rounded headline table. Distinct from2.1; timeouts/resources changed.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":37.3,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"terminal-bench-science-0-1-percent","evidenceKind":"lab_self_report","harnessId":"terminal-bench-science-0-1-percent:anthropic-fable-5-1-card:NzB0YXNrczsgQ2xhdWRlIENvZGUgLS1iYXJlIG1heDsgRmFibGUxMHRyaWFscy90YXNrLCBPcHVzMTI7IEdQVCBDb2RleCBDTEkgbWF4IGZyb20gcHVibGljIGJvYXJkLg","id":"launch-anthropic-fable-5-1-card-23","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"agentic","configuration":"70tasks; Claude Code --bare max; Fable10trials/task, Opus12; GPT Codex CLI max from public board.","family":"Terminal-Bench","locator":"8.7","metric":"percent","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":52.6,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"terminal-bench-science-0-1-percent","evidenceKind":"lab_self_report","harnessId":"terminal-bench-science-0-1-percent:anthropic-fable-5-1-card:NzB0YXNrczsgQ2xhdWRlIENvZGUgLS1iYXJlIG1heDsgRmFibGUxMHRyaWFscy90YXNrLCBPcHVzMTI7IEdQVCBDb2RleCBDTEkgbWF4IGZyb20gcHVibGljIGJvYXJkLg","id":"launch-anthropic-fable-5-1-card-24","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"agentic","configuration":"70tasks; Claude Code --bare max; Fable10trials/task, Opus12; GPT Codex CLI max from public board.","family":"Terminal-Bench","locator":"8.7","metric":"percent","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":24.7,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"terminal-bench-science-0-1-percent","evidenceKind":"lab_self_report","harnessId":"terminal-bench-science-0-1-percent:anthropic-fable-5-1-card:NzB0YXNrczsgQ2xhdWRlIENvZGUgLS1iYXJlIG1heDsgRmFibGUxMHRyaWFscy90YXNrLCBPcHVzMTI7IEdQVCBDb2RleCBDTEkgbWF4IGZyb20gcHVibGljIGJvYXJkLg","id":"launch-anthropic-fable-5-1-card-25","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"agentic","configuration":"70tasks; Claude Code --bare max; Fable10trials/task, Opus12; GPT Codex CLI max from public board.","family":"Terminal-Bench","locator":"8.7","metric":"percent","notes":"","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":29,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"terminal-bench-science-0-1-percent","evidenceKind":"lab_self_report","harnessId":"terminal-bench-science-0-1-percent:anthropic-fable-5-1-card:NzB0YXNrczsgQ2xhdWRlIENvZGUgLS1iYXJlIG1heDsgRmFibGUxMHRyaWFscy90YXNrLCBPcHVzMTI7IEdQVCBDb2RleCBDTEkgbWF4IGZyb20gcHVibGljIGJvYXJkLg","id":"launch-anthropic-fable-5-1-card-26","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"agentic","configuration":"70tasks; Claude Code --bare max; Fable10trials/task, Opus12; GPT Codex CLI max from public board.","family":"Terminal-Bench","locator":"8.7","metric":"percent","notes":"","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":22.4,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"cursorbench-3-2-0","evidenceKind":"lab_self_report","harnessId":"cursorbench-3-2-0:anthropic-fable-5-1-card:Q3Vyc29yIHByb2R1Y3Rpb24gYWdlbnQgaGFybmVzczsgaW5kZXBlbmRlbnRseSBtZWFzdXJlZCBieSBDdXJzb3I7IG1heCBlZmZvcnQu","id":"launch-anthropic-fable-5-1-card-27","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"coding","configuration":"Cursor production agent harness; independently measured by Cursor; max effort.","family":"CursorBench","locator":"8.8","metric":"percent","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":73.4,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"cursorbench-3-2-0","evidenceKind":"lab_self_report","harnessId":"cursorbench-3-2-0:anthropic-fable-5-1-card:Q3Vyc29yIHByb2R1Y3Rpb24gYWdlbnQgaGFybmVzczsgaW5kZXBlbmRlbnRseSBtZWFzdXJlZCBieSBDdXJzb3I7IG1heCBlZmZvcnQu","id":"launch-anthropic-fable-5-1-card-28","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"coding","configuration":"Cursor production agent harness; independently measured by Cursor; max effort.","family":"CursorBench","locator":"8.8","metric":"percent","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":70.5,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"cursorbench-3-2-0","evidenceKind":"lab_self_report","harnessId":"cursorbench-3-2-0:anthropic-fable-5-1-card:Q3Vyc29yIHByb2R1Y3Rpb24gYWdlbnQgaGFybmVzczsgaW5kZXBlbmRlbnRseSBtZWFzdXJlZCBieSBDdXJzb3I7IG1heCBlZmZvcnQu","id":"launch-anthropic-fable-5-1-card-29","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"coding","configuration":"Cursor production agent harness; independently measured by Cursor; max effort.","family":"CursorBench","locator":"8.8","metric":"percent","notes":"","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":70,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"swe-bench-pro-percent-higher","evidenceKind":"lab_self_report","harnessId":"swe-bench-pro-percent-higher:anthropic-fable-5-1-card:QW50aHJvcGljIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IGFkYXB0aXZlIHRoaW5raW5nIGF0IG1heCBlZmZvcnQsIGRlZmF1bHQgc2FtcGxpbmcsIGZpdmUtdHJpYWwgbWVhbiB1bmxlc3Mgc2VjdGlvbiBzcGVjaWZpZXMgb3RoZXJ3aXNlOyBjb250ZXh0IGF0IG1vc3QgMU0uIENvbXBhcmF0b3IgY29uZmlndXJhdGlvbiBmb2xsb3dzIGNpdGVkIHByaW9yIGNhcmQgb3IgYm9hcmQuIEFnZW50IGlkZW50aXR5IGlzIG5vdCBlc3RhYmxpc2hlZCBhcyBtaW5pLXN3ZS1hZ2VudC4","id":"launch-anthropic-fable-5-1-card-3","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"coding","configuration":"Anthropic reported configuration; adaptive thinking at max effort, default sampling, five-trial mean unless section specifies otherwise; context at most 1M. Comparator configuration follows cited prior card or board. Agent identity is not established as mini-swe-agent.","family":"SWE-bench Pro","locator":"Table 8.1.A; section 8.2","metric":"percent","notes":"","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":64.6,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"cursorbench-3-2-0","evidenceKind":"lab_self_report","harnessId":"cursorbench-3-2-0:anthropic-fable-5-1-card:Q3Vyc29yIHByb2R1Y3Rpb24gYWdlbnQgaGFybmVzczsgaW5kZXBlbmRlbnRseSBtZWFzdXJlZCBieSBDdXJzb3I7IG1heCBlZmZvcnQu","id":"launch-anthropic-fable-5-1-card-30","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"coding","configuration":"Cursor production agent harness; independently measured by Cursor; max effort.","family":"CursorBench","locator":"8.8","metric":"percent","notes":"","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":67.2,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"cursorbench-3-2-0","evidenceKind":"lab_self_report","harnessId":"cursorbench-3-2-0:anthropic-fable-5-1-card:Q3Vyc29yIHByb2R1Y3Rpb24gYWdlbnQgaGFybmVzczsgbWVkaXVtIGVmZm9ydDsgJDMuNTMvdGFzay4","id":"launch-anthropic-fable-5-1-card-31","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"coding","configuration":"Cursor production agent harness; medium effort; $3.53/task.","family":"CursorBench","locator":"8.8","metric":"percent","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":68,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"programbench-166-golden-task-subset","evidenceKind":"lab_self_report","harnessId":"programbench-166-golden-task-subset:anthropic-fable-5-1-card:bWluaS1zd2UtYWdlbnQgd2l0aG91dCBzaXgtaG91ciB0aW1lb3V0OyBleGNsdWRlczM0Zmxha3ktcmVmZXJlbmNlIHRhc2tzOyB0ZXN0cyByZXN0cmljdGVkIHRvIHJlZmVyZW5jZS1wYXNzaW5nIHRlc3RzOyB1cCB0bzFNY29udGV4dC4","id":"launch-anthropic-fable-5-1-card-32","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"long_context","configuration":"mini-swe-agent without six-hour timeout; excludes34flaky-reference tasks; tests restricted to reference-passing tests; up to1Mcontext.","family":"ProgramBench","locator":"8.11.1","metric":"percent","notes":"Hidden-test pass rate, not fraction of completely solved programs. Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":87.6,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"programbench-166-golden-task-subset","evidenceKind":"lab_self_report","harnessId":"programbench-166-golden-task-subset:anthropic-fable-5-1-card:bWluaS1zd2UtYWdlbnQgd2l0aG91dCBzaXgtaG91ciB0aW1lb3V0OyBleGNsdWRlczM0Zmxha3ktcmVmZXJlbmNlIHRhc2tzOyB0ZXN0cyByZXN0cmljdGVkIHRvIHJlZmVyZW5jZS1wYXNzaW5nIHRlc3RzOyB1cCB0bzFNY29udGV4dC4","id":"launch-anthropic-fable-5-1-card-33","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"long_context","configuration":"mini-swe-agent without six-hour timeout; excludes34flaky-reference tasks; tests restricted to reference-passing tests; up to1Mcontext.","family":"ProgramBench","locator":"8.11.1","metric":"percent","notes":"Hidden-test pass rate, not fraction of completely solved programs. Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":86.3,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"programbench-166-golden-task-subset","evidenceKind":"lab_self_report","harnessId":"programbench-166-golden-task-subset:anthropic-fable-5-1-card:bWluaS1zd2UtYWdlbnQgd2l0aG91dCBzaXgtaG91ciB0aW1lb3V0OyBleGNsdWRlczM0Zmxha3ktcmVmZXJlbmNlIHRhc2tzOyB0ZXN0cyByZXN0cmljdGVkIHRvIHJlZmVyZW5jZS1wYXNzaW5nIHRlc3RzOyB1cCB0bzFNY29udGV4dC4","id":"launch-anthropic-fable-5-1-card-34","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"long_context","configuration":"mini-swe-agent without six-hour timeout; excludes34flaky-reference tasks; tests restricted to reference-passing tests; up to1Mcontext.","family":"ProgramBench","locator":"8.11.1","metric":"percent","notes":"Hidden-test pass rate, not fraction of completely solved programs.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":85.4,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"humanity-s-last-exam","evidenceKind":"lab_self_report","harnessId":"humanity-s-last-exam:anthropic-fable-5-1-card:RnVsbDI1MDBxdWVzdGlvbnM7IG5vIHRvb2xzOyBhdXRvIHRoaW5raW5nOzFNdG90YWwgdG9rZW4gY2FwOyBubyBjb21wYWN0aW9uOyBPcHVzNC42Z3JhZGVyOyByZXN0cmljdGVkIGZldGNoIGFuZCBjb250YW1pbmF0aW9uIHJldmlldyBmb3IgdG9vbHMu","id":"launch-anthropic-fable-5-1-card-35","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"hard_reasoning","configuration":"Full2500questions; no tools; auto thinking;1Mtotal token cap; no compaction; Opus4.6grader; restricted fetch and contamination review for tools.","family":"Humanity's Last Exam","locator":"Table 8.1.A;8.12.1","metric":"percent","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":60.9,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"humanity-s-last-exam","evidenceKind":"lab_self_report","harnessId":"humanity-s-last-exam:anthropic-fable-5-1-card:RnVsbDI1MDBxdWVzdGlvbnM7IG5vIHRvb2xzOyBhdXRvIHRoaW5raW5nOzFNdG90YWwgdG9rZW4gY2FwOyBubyBjb21wYWN0aW9uOyBPcHVzNC42Z3JhZGVyOyByZXN0cmljdGVkIGZldGNoIGFuZCBjb250YW1pbmF0aW9uIHJldmlldyBmb3IgdG9vbHMu","id":"launch-anthropic-fable-5-1-card-36","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"hard_reasoning","configuration":"Full2500questions; no tools; auto thinking;1Mtotal token cap; no compaction; Opus4.6grader; restricted fetch and contamination review for tools.","family":"Humanity's Last Exam","locator":"Table 8.1.A;8.12.1","metric":"percent","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":57.8,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"humanity-s-last-exam","evidenceKind":"lab_self_report","harnessId":"humanity-s-last-exam:anthropic-fable-5-1-card:RnVsbDI1MDBxdWVzdGlvbnM7IG5vIHRvb2xzOyBhdXRvIHRoaW5raW5nOzFNdG90YWwgdG9rZW4gY2FwOyBubyBjb21wYWN0aW9uOyBPcHVzNC42Z3JhZGVyOyByZXN0cmljdGVkIGZldGNoIGFuZCBjb250YW1pbmF0aW9uIHJldmlldyBmb3IgdG9vbHMu","id":"launch-anthropic-fable-5-1-card-37","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"hard_reasoning","configuration":"Full2500questions; no tools; auto thinking;1Mtotal token cap; no compaction; Opus4.6grader; restricted fetch and contamination review for tools.","family":"Humanity's Last Exam","locator":"Table 8.1.A;8.12.1","metric":"percent","notes":"","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":56.6,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"humanity-s-last-exam","evidenceKind":"lab_self_report","harnessId":"humanity-s-last-exam:anthropic-fable-5-1-card:RnVsbDI1MDBxdWVzdGlvbnM7IHdpdGggdG9vbHM7IGF1dG8gdGhpbmtpbmc7MU10b3RhbCB0b2tlbiBjYXA7IG5vIGNvbXBhY3Rpb247IE9wdXM0LjZncmFkZXI7IHJlc3RyaWN0ZWQgZmV0Y2ggYW5kIGNvbnRhbWluYXRpb24gcmV2aWV3IGZvciB0b29scy4","id":"launch-anthropic-fable-5-1-card-38","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"hard_reasoning","configuration":"Full2500questions; with tools; auto thinking;1Mtotal token cap; no compaction; Opus4.6grader; restricted fetch and contamination review for tools.","family":"Humanity's Last Exam","locator":"Table 8.1.A;8.12.1","metric":"percent","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":65,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"humanity-s-last-exam","evidenceKind":"lab_self_report","harnessId":"humanity-s-last-exam:anthropic-fable-5-1-card:RnVsbDI1MDBxdWVzdGlvbnM7IHdpdGggdG9vbHM7IGF1dG8gdGhpbmtpbmc7MU10b3RhbCB0b2tlbiBjYXA7IG5vIGNvbXBhY3Rpb247IE9wdXM0LjZncmFkZXI7IHJlc3RyaWN0ZWQgZmV0Y2ggYW5kIGNvbnRhbWluYXRpb24gcmV2aWV3IGZvciB0b29scy4","id":"launch-anthropic-fable-5-1-card-39","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"hard_reasoning","configuration":"Full2500questions; with tools; auto thinking;1Mtotal token cap; no compaction; Opus4.6grader; restricted fetch and contamination review for tools.","family":"Humanity's Last Exam","locator":"Table 8.1.A;8.12.1","metric":"percent","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":63.8,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"swe-bench-multilingual-percent","evidenceKind":"lab_self_report","harnessId":"swe-bench-multilingual-percent:anthropic-fable-5-1-card:QW50aHJvcGljIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IGFkYXB0aXZlIHRoaW5raW5nIGF0IG1heCBlZmZvcnQsIGRlZmF1bHQgc2FtcGxpbmcsIGZpdmUtdHJpYWwgbWVhbiB1bmxlc3Mgc2VjdGlvbiBzcGVjaWZpZXMgb3RoZXJ3aXNlOyBjb250ZXh0IGF0IG1vc3QgMU0uIENvbXBhcmF0b3IgY29uZmlndXJhdGlvbiBmb2xsb3dzIGNpdGVkIHByaW9yIGNhcmQgb3IgYm9hcmQuIEFnZW50IGlkZW50aXR5IGlzIG5vdCBlc3RhYmxpc2hlZCBhcyBtaW5pLXN3ZS1hZ2VudC4","id":"launch-anthropic-fable-5-1-card-4","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"coding","configuration":"Anthropic reported configuration; adaptive thinking at max effort, default sampling, five-trial mean unless section specifies otherwise; context at most 1M. Comparator configuration follows cited prior card or board. Agent identity is not established as mini-swe-agent.","family":"SWE-bench Multilingual","locator":"Table 8.1.A; section 8.2","metric":"percent","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":89.1,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"humanity-s-last-exam","evidenceKind":"lab_self_report","harnessId":"humanity-s-last-exam:anthropic-fable-5-1-card:RnVsbDI1MDBxdWVzdGlvbnM7IHdpdGggdG9vbHM7IGF1dG8gdGhpbmtpbmc7MU10b3RhbCB0b2tlbiBjYXA7IG5vIGNvbXBhY3Rpb247IE9wdXM0LjZncmFkZXI7IHJlc3RyaWN0ZWQgZmV0Y2ggYW5kIGNvbnRhbWluYXRpb24gcmV2aWV3IGZvciB0b29scy4","id":"launch-anthropic-fable-5-1-card-40","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"hard_reasoning","configuration":"Full2500questions; with tools; auto thinking;1Mtotal token cap; no compaction; Opus4.6grader; restricted fetch and contamination review for tools.","family":"Humanity's Last Exam","locator":"Table 8.1.A;8.12.1","metric":"percent","notes":"","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":63.6,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"chartography","evidenceKind":"lab_self_report","harnessId":"chartography:anthropic-fable-5-1-card:MTAwdGFza3M7IGFkYXB0aXZlIHRoaW5raW5nIG1heDsgZml2ZSBydW5zOyBubyB0b29sczsgdG9vbHMgY29uZGl0aW9uIGhhcyBjb250YWluZXIsIHN0YW5kYXJkIGxpYnJhcmllcyBhbmQgY3JvcCB0b29sLg","id":"launch-anthropic-fable-5-1-card-41","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"multimodal","configuration":"100tasks; adaptive thinking max; five runs; no tools; tools condition has container, standard libraries and crop tool.","family":"Chartography","locator":"8.14.1","metric":"percent","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":42.6,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"chartography","evidenceKind":"lab_self_report","harnessId":"chartography:anthropic-fable-5-1-card:MTAwdGFza3M7IGFkYXB0aXZlIHRoaW5raW5nIG1heDsgZml2ZSBydW5zOyBubyB0b29sczsgdG9vbHMgY29uZGl0aW9uIGhhcyBjb250YWluZXIsIHN0YW5kYXJkIGxpYnJhcmllcyBhbmQgY3JvcCB0b29sLg","id":"launch-anthropic-fable-5-1-card-42","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"multimodal","configuration":"100tasks; adaptive thinking max; five runs; no tools; tools condition has container, standard libraries and crop tool.","family":"Chartography","locator":"8.14.1","metric":"percent","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":36.6,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"chartography","evidenceKind":"lab_self_report","harnessId":"chartography:anthropic-fable-5-1-card:MTAwdGFza3M7IGFkYXB0aXZlIHRoaW5raW5nIG1heDsgZml2ZSBydW5zOyBubyB0b29sczsgdG9vbHMgY29uZGl0aW9uIGhhcyBjb250YWluZXIsIHN0YW5kYXJkIGxpYnJhcmllcyBhbmQgY3JvcCB0b29sLg","id":"launch-anthropic-fable-5-1-card-43","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"multimodal","configuration":"100tasks; adaptive thinking max; five runs; no tools; tools condition has container, standard libraries and crop tool.","family":"Chartography","locator":"8.14.1","metric":"percent","notes":"","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":29.6,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"chartography","evidenceKind":"lab_self_report","harnessId":"chartography:anthropic-fable-5-1-card:MTAwdGFza3M7IGFkYXB0aXZlIHRoaW5raW5nIG1heDsgZml2ZSBydW5zOyB3aXRoIHRvb2xzOyB0b29scyBjb25kaXRpb24gaGFzIGNvbnRhaW5lciwgc3RhbmRhcmQgbGlicmFyaWVzIGFuZCBjcm9wIHRvb2wu","id":"launch-anthropic-fable-5-1-card-44","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"multimodal","configuration":"100tasks; adaptive thinking max; five runs; with tools; tools condition has container, standard libraries and crop tool.","family":"Chartography","locator":"8.14.1","metric":"percent","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":86.2,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"chartography","evidenceKind":"lab_self_report","harnessId":"chartography:anthropic-fable-5-1-card:MTAwdGFza3M7IGFkYXB0aXZlIHRoaW5raW5nIG1heDsgZml2ZSBydW5zOyB3aXRoIHRvb2xzOyB0b29scyBjb25kaXRpb24gaGFzIGNvbnRhaW5lciwgc3RhbmRhcmQgbGlicmFyaWVzIGFuZCBjcm9wIHRvb2wu","id":"launch-anthropic-fable-5-1-card-45","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"multimodal","configuration":"100tasks; adaptive thinking max; five runs; with tools; tools condition has container, standard libraries and crop tool.","family":"Chartography","locator":"8.14.1","metric":"percent","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":84.2,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"chartography","evidenceKind":"lab_self_report","harnessId":"chartography:anthropic-fable-5-1-card:MTAwdGFza3M7IGFkYXB0aXZlIHRoaW5raW5nIG1heDsgZml2ZSBydW5zOyB3aXRoIHRvb2xzOyB0b29scyBjb25kaXRpb24gaGFzIGNvbnRhaW5lciwgc3RhbmRhcmQgbGlicmFyaWVzIGFuZCBjcm9wIHRvb2wu","id":"launch-anthropic-fable-5-1-card-46","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"multimodal","configuration":"100tasks; adaptive thinking max; five runs; with tools; tools condition has container, standard libraries and crop tool.","family":"Chartography","locator":"8.14.1","metric":"percent","notes":"","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":83,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"benchcad-vision2code-1000-file-subset","evidenceKind":"lab_self_report","harnessId":"benchcad-vision2code-1000-file-subset:anthropic-fable-5-1-card:UmFuZG9tMTAwMCBvZjE3OTAwZmlsZXM7IGZpdmUgcnVuczsgYWRhcHRpdmUgdGhpbmtpbmcgbWF4OyBubyB0b29sczsgY29ycmVjdGVkIGNhbWVyYSBwcm9tcHQsIHJhdyBzaGFwZXMgYWNjZXB0ZWQsIGxhc3QgY29kZSBmZW5jZSBwYXJzZWQu","id":"launch-anthropic-fable-5-1-card-47","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"multimodal","configuration":"Random1000 of17900files; five runs; adaptive thinking max; no tools; corrected camera prompt, raw shapes accepted, last code fence parsed.","family":"BenchCAD","locator":"8.14.2","metric":"voxel IoU","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":0.437,"scoreUnit":"index","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"benchcad-vision2code-1000-file-subset","evidenceKind":"lab_self_report","harnessId":"benchcad-vision2code-1000-file-subset:anthropic-fable-5-1-card:UmFuZG9tMTAwMCBvZjE3OTAwZmlsZXM7IGZpdmUgcnVuczsgYWRhcHRpdmUgdGhpbmtpbmcgbWF4OyBubyB0b29sczsgY29ycmVjdGVkIGNhbWVyYSBwcm9tcHQsIHJhdyBzaGFwZXMgYWNjZXB0ZWQsIGxhc3QgY29kZSBmZW5jZSBwYXJzZWQu","id":"launch-anthropic-fable-5-1-card-48","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"multimodal","configuration":"Random1000 of17900files; five runs; adaptive thinking max; no tools; corrected camera prompt, raw shapes accepted, last code fence parsed.","family":"BenchCAD","locator":"8.14.2","metric":"voxel IoU","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":0.376,"scoreUnit":"index","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"benchcad-vision2code-1000-file-subset","evidenceKind":"lab_self_report","harnessId":"benchcad-vision2code-1000-file-subset:anthropic-fable-5-1-card:UmFuZG9tMTAwMCBvZjE3OTAwZmlsZXM7IGZpdmUgcnVuczsgYWRhcHRpdmUgdGhpbmtpbmcgbWF4OyBubyB0b29sczsgY29ycmVjdGVkIGNhbWVyYSBwcm9tcHQsIHJhdyBzaGFwZXMgYWNjZXB0ZWQsIGxhc3QgY29kZSBmZW5jZSBwYXJzZWQu","id":"launch-anthropic-fable-5-1-card-49","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"multimodal","configuration":"Random1000 of17900files; five runs; adaptive thinking max; no tools; corrected camera prompt, raw shapes accepted, last code fence parsed.","family":"BenchCAD","locator":"8.14.2","metric":"voxel IoU","notes":"","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":0.366,"scoreUnit":"index","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"swe-bench-multilingual-percent","evidenceKind":"lab_self_report","harnessId":"swe-bench-multilingual-percent:anthropic-fable-5-1-card:QW50aHJvcGljIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IGFkYXB0aXZlIHRoaW5raW5nIGF0IG1heCBlZmZvcnQsIGRlZmF1bHQgc2FtcGxpbmcsIGZpdmUtdHJpYWwgbWVhbiB1bmxlc3Mgc2VjdGlvbiBzcGVjaWZpZXMgb3RoZXJ3aXNlOyBjb250ZXh0IGF0IG1vc3QgMU0uIENvbXBhcmF0b3IgY29uZmlndXJhdGlvbiBmb2xsb3dzIGNpdGVkIHByaW9yIGNhcmQgb3IgYm9hcmQuIEFnZW50IGlkZW50aXR5IGlzIG5vdCBlc3RhYmxpc2hlZCBhcyBtaW5pLXN3ZS1hZ2VudC4","id":"launch-anthropic-fable-5-1-card-5","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"coding","configuration":"Anthropic reported configuration; adaptive thinking at max effort, default sampling, five-trial mean unless section specifies otherwise; context at most 1M. Comparator configuration follows cited prior card or board. Agent identity is not established as mini-swe-agent.","family":"SWE-bench Multilingual","locator":"Table 8.1.A; section 8.2","metric":"percent","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":86.6,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"benchcad-vision2code-1000-file-subset","evidenceKind":"lab_self_report","harnessId":"benchcad-vision2code-1000-file-subset:anthropic-fable-5-1-card:UmFuZG9tMTAwMCBvZjE3OTAwZmlsZXM7IGZpdmUgcnVuczsgYWRhcHRpdmUgdGhpbmtpbmcgbWF4OyB3aXRoIHRvb2xzOyBjb3JyZWN0ZWQgY2FtZXJhIHByb21wdCwgcmF3IHNoYXBlcyBhY2NlcHRlZCwgbGFzdCBjb2RlIGZlbmNlIHBhcnNlZC4","id":"launch-anthropic-fable-5-1-card-50","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"multimodal","configuration":"Random1000 of17900files; five runs; adaptive thinking max; with tools; corrected camera prompt, raw shapes accepted, last code fence parsed.","family":"BenchCAD","locator":"8.14.2","metric":"voxel IoU","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":0.843,"scoreUnit":"index","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"benchcad-vision2code-1000-file-subset","evidenceKind":"lab_self_report","harnessId":"benchcad-vision2code-1000-file-subset:anthropic-fable-5-1-card:UmFuZG9tMTAwMCBvZjE3OTAwZmlsZXM7IGZpdmUgcnVuczsgYWRhcHRpdmUgdGhpbmtpbmcgbWF4OyB3aXRoIHRvb2xzOyBjb3JyZWN0ZWQgY2FtZXJhIHByb21wdCwgcmF3IHNoYXBlcyBhY2NlcHRlZCwgbGFzdCBjb2RlIGZlbmNlIHBhcnNlZC4","id":"launch-anthropic-fable-5-1-card-51","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"multimodal","configuration":"Random1000 of17900files; five runs; adaptive thinking max; with tools; corrected camera prompt, raw shapes accepted, last code fence parsed.","family":"BenchCAD","locator":"8.14.2","metric":"voxel IoU","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":0.675,"scoreUnit":"index","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"benchcad-vision2code-1000-file-subset","evidenceKind":"lab_self_report","harnessId":"benchcad-vision2code-1000-file-subset:anthropic-fable-5-1-card:UmFuZG9tMTAwMCBvZjE3OTAwZmlsZXM7IGZpdmUgcnVuczsgYWRhcHRpdmUgdGhpbmtpbmcgbWF4OyB3aXRoIHRvb2xzOyBjb3JyZWN0ZWQgY2FtZXJhIHByb21wdCwgcmF3IHNoYXBlcyBhY2NlcHRlZCwgbGFzdCBjb2RlIGZlbmNlIHBhcnNlZC4","id":"launch-anthropic-fable-5-1-card-52","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"multimodal","configuration":"Random1000 of17900files; five runs; adaptive thinking max; with tools; corrected camera prompt, raw shapes accepted, last code fence parsed.","family":"BenchCAD","locator":"8.14.2","metric":"voxel IoU","notes":"","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":0.821,"scoreUnit":"index","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"osworld-2-0-august2026-task-release","evidenceKind":"lab_self_report","harnessId":"osworld-2-0-august2026-task-release:anthropic-fable-5-1-card:cGFydGlhbCBwYXNzQDE7MTA4dGFza3M7IGZpdmUgcnVuczsxMDgwcDs1MDBzdGVwczsgbWF4IGVmZm9ydDsgT3B1czQuOGdyYWRlcjsgdGFzayBmaXhlczsgRmFibGUgc2FmZXR5IGludGVydmVudGlvbnMgc2NvcmUgemVyby4","id":"launch-anthropic-fable-5-1-card-53","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"agentic","configuration":"partial pass@1;108tasks; five runs;1080p;500steps; max effort; Opus4.8grader; task fixes; Fable safety interventions score zero.","family":"OSWorld","locator":"8.14.3","metric":"percent","notes":"Same-condition reruns; supersedes earlier OSWorld2results and incompatible with previous task files. Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":77.9,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"osworld-2-0-august2026-task-release","evidenceKind":"lab_self_report","harnessId":"osworld-2-0-august2026-task-release:anthropic-fable-5-1-card:cGFydGlhbCBwYXNzQDE7MTA4dGFza3M7IGZpdmUgcnVuczsxMDgwcDs1MDBzdGVwczsgbWF4IGVmZm9ydDsgT3B1czQuOGdyYWRlcjsgdGFzayBmaXhlczsgRmFibGUgc2FmZXR5IGludGVydmVudGlvbnMgc2NvcmUgemVyby4","id":"launch-anthropic-fable-5-1-card-54","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"agentic","configuration":"partial pass@1;108tasks; five runs;1080p;500steps; max effort; Opus4.8grader; task fixes; Fable safety interventions score zero.","family":"OSWorld","locator":"8.14.3","metric":"percent","notes":"Same-condition reruns; supersedes earlier OSWorld2results and incompatible with previous task files. Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":72.9,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"osworld-2-0-august2026-task-release","evidenceKind":"lab_self_report","harnessId":"osworld-2-0-august2026-task-release:anthropic-fable-5-1-card:cGFydGlhbCBwYXNzQDE7MTA4dGFza3M7IGZpdmUgcnVuczsxMDgwcDs1MDBzdGVwczsgbWF4IGVmZm9ydDsgT3B1czQuOGdyYWRlcjsgdGFzayBmaXhlczsgRmFibGUgc2FmZXR5IGludGVydmVudGlvbnMgc2NvcmUgemVyby4","id":"launch-anthropic-fable-5-1-card-55","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"agentic","configuration":"partial pass@1;108tasks; five runs;1080p;500steps; max effort; Opus4.8grader; task fixes; Fable safety interventions score zero.","family":"OSWorld","locator":"8.14.3","metric":"percent","notes":"Same-condition reruns; supersedes earlier OSWorld2results and incompatible with previous task files.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":75.4,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"osworld-2-0-august2026-task-release","evidenceKind":"lab_self_report","harnessId":"osworld-2-0-august2026-task-release:anthropic-fable-5-1-card:c3RyaWN0IHBhc3NAMTsxMDh0YXNrczsgZml2ZSBydW5zOzEwODBwOzUwMHN0ZXBzOyBtYXggZWZmb3J0OyBPcHVzNC44Z3JhZGVyOyB0YXNrIGZpeGVzOyBGYWJsZSBzYWZldHkgaW50ZXJ2ZW50aW9ucyBzY29yZSB6ZXJvLg","id":"launch-anthropic-fable-5-1-card-56","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"agentic","configuration":"strict pass@1;108tasks; five runs;1080p;500steps; max effort; Opus4.8grader; task fixes; Fable safety interventions score zero.","family":"OSWorld","locator":"8.14.3","metric":"percent","notes":"Same-condition reruns; supersedes earlier OSWorld2results and incompatible with previous task files. Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":41.7,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"osworld-2-0-august2026-task-release","evidenceKind":"lab_self_report","harnessId":"osworld-2-0-august2026-task-release:anthropic-fable-5-1-card:c3RyaWN0IHBhc3NAMTsxMDh0YXNrczsgZml2ZSBydW5zOzEwODBwOzUwMHN0ZXBzOyBtYXggZWZmb3J0OyBPcHVzNC44Z3JhZGVyOyB0YXNrIGZpeGVzOyBGYWJsZSBzYWZldHkgaW50ZXJ2ZW50aW9ucyBzY29yZSB6ZXJvLg","id":"launch-anthropic-fable-5-1-card-57","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"agentic","configuration":"strict pass@1;108tasks; five runs;1080p;500steps; max effort; Opus4.8grader; task fixes; Fable safety interventions score zero.","family":"OSWorld","locator":"8.14.3","metric":"percent","notes":"Same-condition reruns; supersedes earlier OSWorld2results and incompatible with previous task files. Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":36.1,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"osworld-2-0-august2026-task-release","evidenceKind":"lab_self_report","harnessId":"osworld-2-0-august2026-task-release:anthropic-fable-5-1-card:c3RyaWN0IHBhc3NAMTsxMDh0YXNrczsgZml2ZSBydW5zOzEwODBwOzUwMHN0ZXBzOyBtYXggZWZmb3J0OyBPcHVzNC44Z3JhZGVyOyB0YXNrIGZpeGVzOyBGYWJsZSBzYWZldHkgaW50ZXJ2ZW50aW9ucyBzY29yZSB6ZXJvLg","id":"launch-anthropic-fable-5-1-card-58","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"agentic","configuration":"strict pass@1;108tasks; five runs;1080p;500steps; max effort; Opus4.8grader; task fixes; Fable safety interventions score zero.","family":"OSWorld","locator":"8.14.3","metric":"percent","notes":"Same-condition reruns; supersedes earlier OSWorld2results and incompatible with previous task files.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":39.6,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"officeqa","evidenceKind":"lab_self_report","harnessId":"officeqa:anthropic-fable-5-1-card:RXh0cmFjdGVkLXRleHQgVHJlYXN1cnkgY29ycHVzIGluIHNhbmRib3g7IGNvZGUgZXhlY3V0aW9uOyBwcm9kdWN0aW9uIE1lc3NhZ2VzIEFQSSB3aXRoIHNhZmVndWFyZHMvZmFsbGJhY2s7MTI4a291dHB1dCBjYXAuIFBybzEzM3F1ZXN0aW9ucy4","id":"launch-anthropic-fable-5-1-card-59","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"knowledge","configuration":"Extracted-text Treasury corpus in sandbox; code execution; production Messages API with safeguards/fallback;128koutput cap. Pro133questions.","family":"OfficeQA","locator":"8.15.1","metric":"percent","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":80.2,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"swe-bench-multilingual-percent","evidenceKind":"lab_self_report","harnessId":"swe-bench-multilingual-percent:anthropic-fable-5-1-card:QW50aHJvcGljIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IGFkYXB0aXZlIHRoaW5raW5nIGF0IG1heCBlZmZvcnQsIGRlZmF1bHQgc2FtcGxpbmcsIGZpdmUtdHJpYWwgbWVhbiB1bmxlc3Mgc2VjdGlvbiBzcGVjaWZpZXMgb3RoZXJ3aXNlOyBjb250ZXh0IGF0IG1vc3QgMU0uIENvbXBhcmF0b3IgY29uZmlndXJhdGlvbiBmb2xsb3dzIGNpdGVkIHByaW9yIGNhcmQgb3IgYm9hcmQuIEFnZW50IGlkZW50aXR5IGlzIG5vdCBlc3RhYmxpc2hlZCBhcyBtaW5pLXN3ZS1hZ2VudC4","id":"launch-anthropic-fable-5-1-card-6","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"coding","configuration":"Anthropic reported configuration; adaptive thinking at max effort, default sampling, five-trial mean unless section specifies otherwise; context at most 1M. Comparator configuration follows cited prior card or board. Agent identity is not established as mini-swe-agent.","family":"SWE-bench Multilingual","locator":"Table 8.1.A; section 8.2","metric":"percent","notes":"","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":89.5,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"officeqa","evidenceKind":"lab_self_report","harnessId":"officeqa:anthropic-fable-5-1-card:RXh0cmFjdGVkLXRleHQgVHJlYXN1cnkgY29ycHVzIGluIHNhbmRib3g7IGNvZGUgZXhlY3V0aW9uOyBwcm9kdWN0aW9uIE1lc3NhZ2VzIEFQSSB3aXRoIHNhZmVndWFyZHMvZmFsbGJhY2s7MTI4a291dHB1dCBjYXAuIFBybzEzM3F1ZXN0aW9ucy4","id":"launch-anthropic-fable-5-1-card-60","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"knowledge","configuration":"Extracted-text Treasury corpus in sandbox; code execution; production Messages API with safeguards/fallback;128koutput cap. Pro133questions.","family":"OfficeQA","locator":"8.15.1","metric":"percent","notes":"","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":78.1,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"officeqa-pro","evidenceKind":"lab_self_report","harnessId":"officeqa-pro:anthropic-fable-5-1-card:RXh0cmFjdGVkLXRleHQgVHJlYXN1cnkgY29ycHVzIGluIHNhbmRib3g7IGNvZGUgZXhlY3V0aW9uOyBwcm9kdWN0aW9uIE1lc3NhZ2VzIEFQSSB3aXRoIHNhZmVndWFyZHMvZmFsbGJhY2s7MTI4a291dHB1dCBjYXAuIFBybzEzM3F1ZXN0aW9ucy4","id":"launch-anthropic-fable-5-1-card-61","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"knowledge","configuration":"Extracted-text Treasury corpus in sandbox; code execution; production Messages API with safeguards/fallback;128koutput cap. Pro133questions.","family":"OfficeQA Pro","locator":"8.15.1","metric":"percent","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":69,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"officeqa-pro","evidenceKind":"lab_self_report","harnessId":"officeqa-pro:anthropic-fable-5-1-card:RXh0cmFjdGVkLXRleHQgVHJlYXN1cnkgY29ycHVzIGluIHNhbmRib3g7IGNvZGUgZXhlY3V0aW9uOyBwcm9kdWN0aW9uIE1lc3NhZ2VzIEFQSSB3aXRoIHNhZmVndWFyZHMvZmFsbGJhY2s7MTI4a291dHB1dCBjYXAuIFBybzEzM3F1ZXN0aW9ucy4","id":"launch-anthropic-fable-5-1-card-62","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"knowledge","configuration":"Extracted-text Treasury corpus in sandbox; code execution; production Messages API with safeguards/fallback;128koutput cap. Pro133questions.","family":"OfficeQA Pro","locator":"8.15.1","metric":"percent","notes":"","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":66.9,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"officeqa-pro","evidenceKind":"lab_self_report","harnessId":"officeqa-pro:anthropic-fable-5-1-card:RGF0YWJyaWNrcyBldmFsdWF0aW9uIHJlYWRpbmcgZG9jdW1lbnRzIGFzIGltYWdlczsgZGlmZmVycyBmcm9tIGV4dHJhY3RlZC10ZXh0IGhhcm5lc3Mu","id":"launch-anthropic-fable-5-1-card-63","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"knowledge","configuration":"Databricks evaluation reading documents as images; differs from extracted-text harness.","family":"OfficeQA Pro","locator":"8.15.1","metric":"percent","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":57.9,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"legal-agent-benchmark-1235-task-public-subset","evidenceKind":"lab_self_report","harnessId":"legal-agent-benchmark-1235-task-public-subset:anthropic-fable-5-1-card:YWxsLXBhc3M7IGZpdmUgcnVuczsgYWRhcHRpdmUgbWF4OyBpbnRlcm5hbCBiYXNoL1B5dGhvbiBoYXJuZXNzLCBTb25uZXQ0LjZqdWRnZTsxNmRlZmVjdGl2ZSB0YXNrcyBleGNsdWRlZDsgcHJvZHVjdGlvbiBzYWZlZ3VhcmRzL2ZhbGxiYWNrLg","id":"launch-anthropic-fable-5-1-card-64","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"agentic","configuration":"all-pass; five runs; adaptive max; internal bash/Python harness, Sonnet4.6judge;16defective tasks excluded; production safeguards/fallback.","family":"Legal Agent Benchmark","locator":"8.15.2","metric":"percent","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":19.09,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"legal-agent-benchmark-1235-task-public-subset","evidenceKind":"lab_self_report","harnessId":"legal-agent-benchmark-1235-task-public-subset:anthropic-fable-5-1-card:Y3JpdGVyaW9uLXBhc3M7IGZpdmUgcnVuczsgYWRhcHRpdmUgbWF4OyBpbnRlcm5hbCBiYXNoL1B5dGhvbiBoYXJuZXNzLCBTb25uZXQ0LjZqdWRnZTsxNmRlZmVjdGl2ZSB0YXNrcyBleGNsdWRlZDsgcHJvZHVjdGlvbiBzYWZlZ3VhcmRzL2ZhbGxiYWNrLg","id":"launch-anthropic-fable-5-1-card-65","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"agentic","configuration":"criterion-pass; five runs; adaptive max; internal bash/Python harness, Sonnet4.6judge;16defective tasks excluded; production safeguards/fallback.","family":"Legal Agent Benchmark","locator":"8.15.2","metric":"percent","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":90.81,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"legal-agent-benchmark-120-task-held-out-subset","evidenceKind":"lab_self_report","harnessId":"legal-agent-benchmark-120-task-held-out-subset:anthropic-fable-5-1-card:YWxsLXBhc3M7IEFydGlmaWNpYWwgQW5hbHlzaXMgaGFybmVzczsgeGhpZ2ggZWZmb3J0Lg","id":"launch-anthropic-fable-5-1-card-66","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"agentic","configuration":"all-pass; Artificial Analysis harness; xhigh effort.","family":"Legal Agent Benchmark","locator":"8.15.2","metric":"percent","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":16.7,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"legal-agent-benchmark-120-task-held-out-subset","evidenceKind":"lab_self_report","harnessId":"legal-agent-benchmark-120-task-held-out-subset:anthropic-fable-5-1-card:Y3JpdGVyaW9uLXBhc3M7IEFydGlmaWNpYWwgQW5hbHlzaXMgaGFybmVzczsgeGhpZ2ggZWZmb3J0Lg","id":"launch-anthropic-fable-5-1-card-67","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"agentic","configuration":"criterion-pass; Artificial Analysis harness; xhigh effort.","family":"Legal Agent Benchmark","locator":"8.15.2","metric":"percent","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":93.3,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"gdpval-aa-2","evidenceKind":"lab_self_report","harnessId":"gdpval-aa-2:anthropic-fable-5-1-card:QXJ0aWZpY2lhbCBBbmFseXNpcyBpbmRlcGVuZGVudCBhZ2VudGljIHNoZWxsL3dlYiBldmFsdWF0aW9uOzIyMEdEUHZhbGdvbGR0YXNrczsgYmxpbmQgcGFpcndpc2UgRWxvOyBtYXggZWZmb3J0IGZvciBDbGF1ZGUu","id":"launch-anthropic-fable-5-1-card-68","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"human_pref","configuration":"Artificial Analysis independent agentic shell/web evaluation;220GDPvalgoldtasks; blind pairwise Elo; max effort for Claude.","family":"GDPval-AA","locator":"Table8.1.A;8.15.3","metric":"Elo","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":1853,"scoreUnit":"elo","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"gdpval-aa-2","evidenceKind":"lab_self_report","harnessId":"gdpval-aa-2:anthropic-fable-5-1-card:QXJ0aWZpY2lhbCBBbmFseXNpcyBpbmRlcGVuZGVudCBhZ2VudGljIHNoZWxsL3dlYiBldmFsdWF0aW9uOzIyMEdEUHZhbGdvbGR0YXNrczsgYmxpbmQgcGFpcndpc2UgRWxvOyBtYXggZWZmb3J0IGZvciBDbGF1ZGUu","id":"launch-anthropic-fable-5-1-card-69","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"human_pref","configuration":"Artificial Analysis independent agentic shell/web evaluation;220GDPvalgoldtasks; blind pairwise Elo; max effort for Claude.","family":"GDPval-AA","locator":"Table8.1.A;8.15.3","metric":"Elo","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":1723,"scoreUnit":"elo","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"swe-bench-multimodal","evidenceKind":"lab_self_report","harnessId":"swe-bench-multimodal:anthropic-fable-5-1-card:QW50aHJvcGljIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IGFkYXB0aXZlIHRoaW5raW5nIGF0IG1heCBlZmZvcnQsIGRlZmF1bHQgc2FtcGxpbmcsIGZpdmUtdHJpYWwgbWVhbiB1bmxlc3Mgc2VjdGlvbiBzcGVjaWZpZXMgb3RoZXJ3aXNlOyBjb250ZXh0IGF0IG1vc3QgMU0uIENvbXBhcmF0b3IgY29uZmlndXJhdGlvbiBmb2xsb3dzIGNpdGVkIHByaW9yIGNhcmQgb3IgYm9hcmQuIEFnZW50IGlkZW50aXR5IGlzIG5vdCBlc3RhYmxpc2hlZCBhcyBtaW5pLXN3ZS1hZ2VudC4","id":"launch-anthropic-fable-5-1-card-7","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"coding","configuration":"Anthropic reported configuration; adaptive thinking at max effort, default sampling, five-trial mean unless section specifies otherwise; context at most 1M. Comparator configuration follows cited prior card or board. Agent identity is not established as mini-swe-agent.","family":"SWE-bench Multimodal","locator":"Table 8.1.A; section 8.2","metric":"percent","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":54.7,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"gdpval-aa-2","evidenceKind":"lab_self_report","harnessId":"gdpval-aa-2:anthropic-fable-5-1-card:QXJ0aWZpY2lhbCBBbmFseXNpcyBpbmRlcGVuZGVudCBhZ2VudGljIHNoZWxsL3dlYiBldmFsdWF0aW9uOzIyMEdEUHZhbGdvbGR0YXNrczsgYmxpbmQgcGFpcndpc2UgRWxvOyBtYXggZWZmb3J0IGZvciBDbGF1ZGUu","id":"launch-anthropic-fable-5-1-card-70","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"human_pref","configuration":"Artificial Analysis independent agentic shell/web evaluation;220GDPvalgoldtasks; blind pairwise Elo; max effort for Claude.","family":"GDPval-AA","locator":"Table8.1.A;8.15.3","metric":"Elo","notes":"","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":1824,"scoreUnit":"elo","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"gdpval-aa-2","evidenceKind":"lab_self_report","harnessId":"gdpval-aa-2:anthropic-fable-5-1-card:QXJ0aWZpY2lhbCBBbmFseXNpcyBpbmRlcGVuZGVudCBhZ2VudGljIHNoZWxsL3dlYiBldmFsdWF0aW9uOzIyMEdEUHZhbGdvbGR0YXNrczsgYmxpbmQgcGFpcndpc2UgRWxvOyBtYXggZWZmb3J0IGZvciBDbGF1ZGUu","id":"launch-anthropic-fable-5-1-card-71","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"human_pref","configuration":"Artificial Analysis independent agentic shell/web evaluation;220GDPvalgoldtasks; blind pairwise Elo; max effort for Claude.","family":"GDPval-AA","locator":"Table8.1.A;8.15.3","metric":"Elo","notes":"","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":1711,"scoreUnit":"elo","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"gdpval-aa-2","evidenceKind":"lab_self_report","harnessId":"gdpval-aa-2:anthropic-fable-5-1-card:QXJ0aWZpY2lhbCBBbmFseXNpczsgeGhpZ2ggZWZmb3J0OyBzYW1lIHJlbGVhc2UgYm9hcmQgc25hcHNob3Qu","id":"launch-anthropic-fable-5-1-card-72","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"human_pref","configuration":"Artificial Analysis; xhigh effort; same release board snapshot.","family":"GDPval-AA","locator":"8.15.3","metric":"Elo","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":1835,"scoreUnit":"elo","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"aa-briefcase","evidenceKind":"lab_self_report","harnessId":"aa-briefcase:anthropic-fable-5-1-card:QXJ0aWZpY2lhbCBBbmFseXNpcyBsb25nLWhvcml6b24ga25vd2xlZGdlIHByb2plY3RzOyBydWJyaWMgYW5kIHBhbmVsIHBhaXJ3aXNlIGp1ZGdpbmc7IENsYXVkZSBtYXggZWZmb3J0Lg","id":"launch-anthropic-fable-5-1-card-73","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"human_pref","configuration":"Artificial Analysis long-horizon knowledge projects; rubric and panel pairwise judging; Claude max effort.","family":"AA-Briefcase","locator":"Table8.1.A;8.15.4","metric":"Elo","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":1694,"scoreUnit":"elo","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"aa-briefcase","evidenceKind":"lab_self_report","harnessId":"aa-briefcase:anthropic-fable-5-1-card:QXJ0aWZpY2lhbCBBbmFseXNpcyBsb25nLWhvcml6b24ga25vd2xlZGdlIHByb2plY3RzOyBydWJyaWMgYW5kIHBhbmVsIHBhaXJ3aXNlIGp1ZGdpbmc7IENsYXVkZSBtYXggZWZmb3J0Lg","id":"launch-anthropic-fable-5-1-card-74","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"human_pref","configuration":"Artificial Analysis long-horizon knowledge projects; rubric and panel pairwise judging; Claude max effort.","family":"AA-Briefcase","locator":"Table8.1.A;8.15.4","metric":"Elo","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":1572,"scoreUnit":"elo","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"aa-briefcase","evidenceKind":"lab_self_report","harnessId":"aa-briefcase:anthropic-fable-5-1-card:QXJ0aWZpY2lhbCBBbmFseXNpcyBsb25nLWhvcml6b24ga25vd2xlZGdlIHByb2plY3RzOyBydWJyaWMgYW5kIHBhbmVsIHBhaXJ3aXNlIGp1ZGdpbmc7IENsYXVkZSBtYXggZWZmb3J0Lg","id":"launch-anthropic-fable-5-1-card-75","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"human_pref","configuration":"Artificial Analysis long-horizon knowledge projects; rubric and panel pairwise judging; Claude max effort.","family":"AA-Briefcase","locator":"Table8.1.A;8.15.4","metric":"Elo","notes":"","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":1685,"scoreUnit":"elo","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"aa-briefcase","evidenceKind":"lab_self_report","harnessId":"aa-briefcase:anthropic-fable-5-1-card:QXJ0aWZpY2lhbCBBbmFseXNpcyBsb25nLWhvcml6b24ga25vd2xlZGdlIHByb2plY3RzOyBydWJyaWMgYW5kIHBhbmVsIHBhaXJ3aXNlIGp1ZGdpbmc7IENsYXVkZSBtYXggZWZmb3J0Lg","id":"launch-anthropic-fable-5-1-card-76","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"human_pref","configuration":"Artificial Analysis long-horizon knowledge projects; rubric and panel pairwise judging; Claude max effort.","family":"AA-Briefcase","locator":"Table8.1.A;8.15.4","metric":"Elo","notes":"","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":1502,"scoreUnit":"elo","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"aa-briefcase","evidenceKind":"lab_self_report","harnessId":"aa-briefcase:anthropic-fable-5-1-card:QXJ0aWZpY2lhbCBBbmFseXNpczsgeGhpZ2ggZWZmb3J0Lg","id":"launch-anthropic-fable-5-1-card-77","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"human_pref","configuration":"Artificial Analysis; xhigh effort.","family":"AA-Briefcase","locator":"8.15.4","metric":"Elo","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":1686,"scoreUnit":"elo","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"aa-briefcase","evidenceKind":"lab_self_report","harnessId":"aa-briefcase:anthropic-fable-5-1-card:QXJ0aWZpY2lhbCBBbmFseXNpczsgaGlnaCBlZmZvcnQu","id":"launch-anthropic-fable-5-1-card-78","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"human_pref","configuration":"Artificial Analysis; high effort.","family":"AA-Briefcase","locator":"8.15.4","metric":"Elo","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":1611,"scoreUnit":"elo","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"aa-briefcase-percent","evidenceKind":"lab_self_report","harnessId":"aa-briefcase-percent:anthropic-fable-5-1-card:QXJ0aWZpY2lhbCBBbmFseXNpczsgbWF4IGVmZm9ydDsgY29tcG9uZW50IHJ1YnJpYyBwYXNzLg","id":"launch-anthropic-fable-5-1-card-79","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"human_pref","configuration":"Artificial Analysis; max effort; component rubric pass.","family":"AA-Briefcase","locator":"8.15.4","metric":"percent","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":61.5,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"swe-bench-multimodal","evidenceKind":"lab_self_report","harnessId":"swe-bench-multimodal:anthropic-fable-5-1-card:QW50aHJvcGljIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IGFkYXB0aXZlIHRoaW5raW5nIGF0IG1heCBlZmZvcnQsIGRlZmF1bHQgc2FtcGxpbmcsIGZpdmUtdHJpYWwgbWVhbiB1bmxlc3Mgc2VjdGlvbiBzcGVjaWZpZXMgb3RoZXJ3aXNlOyBjb250ZXh0IGF0IG1vc3QgMU0uIENvbXBhcmF0b3IgY29uZmlndXJhdGlvbiBmb2xsb3dzIGNpdGVkIHByaW9yIGNhcmQgb3IgYm9hcmQuIEFnZW50IGlkZW50aXR5IGlzIG5vdCBlc3RhYmxpc2hlZCBhcyBtaW5pLXN3ZS1hZ2VudC4","id":"launch-anthropic-fable-5-1-card-8","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"coding","configuration":"Anthropic reported configuration; adaptive thinking at max effort, default sampling, five-trial mean unless section specifies otherwise; context at most 1M. Comparator configuration follows cited prior card or board. Agent identity is not established as mini-swe-agent.","family":"SWE-bench Multimodal","locator":"Table 8.1.A; section 8.2","metric":"percent","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":54.1,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"aa-briefcase-percent","evidenceKind":"lab_self_report","harnessId":"aa-briefcase-percent:anthropic-fable-5-1-card:QXJ0aWZpY2lhbCBBbmFseXNpczsgbWF4IGVmZm9ydDsgY29tcG9uZW50IHJ1YnJpYyBwYXNzLg","id":"launch-anthropic-fable-5-1-card-80","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"human_pref","configuration":"Artificial Analysis; max effort; component rubric pass.","family":"AA-Briefcase","locator":"8.15.4","metric":"percent","notes":"","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":57.2,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"aa-briefcase-rating","evidenceKind":"lab_self_report","harnessId":"aa-briefcase-rating:anthropic-fable-5-1-card:QXJ0aWZpY2lhbCBBbmFseXNpczsgbWF4IGVmZm9ydDsgY29tcG9uZW50IGFuYWx5dGljYWwgcXVhbGl0eS4","id":"launch-anthropic-fable-5-1-card-81","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"human_pref","configuration":"Artificial Analysis; max effort; component analytical quality.","family":"AA-Briefcase","locator":"8.15.4","metric":"rating","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":2025,"scoreUnit":"index","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"aa-briefcase-rating","evidenceKind":"lab_self_report","harnessId":"aa-briefcase-rating:anthropic-fable-5-1-card:QXJ0aWZpY2lhbCBBbmFseXNpczsgbWF4IGVmZm9ydDsgY29tcG9uZW50IGFuYWx5dGljYWwgcXVhbGl0eS4","id":"launch-anthropic-fable-5-1-card-82","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"human_pref","configuration":"Artificial Analysis; max effort; component analytical quality.","family":"AA-Briefcase","locator":"8.15.4","metric":"rating","notes":"","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":1980,"scoreUnit":"index","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"aa-briefcase-rating","evidenceKind":"lab_self_report","harnessId":"aa-briefcase-rating:anthropic-fable-5-1-card:QXJ0aWZpY2lhbCBBbmFseXNpczsgbWF4IGVmZm9ydDsgY29tcG9uZW50IHByZXNlbnRhdGlvbi4","id":"launch-anthropic-fable-5-1-card-83","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"human_pref","configuration":"Artificial Analysis; max effort; component presentation.","family":"AA-Briefcase","locator":"8.15.4","metric":"rating","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":1495,"scoreUnit":"index","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"aa-briefcase-rating","evidenceKind":"lab_self_report","harnessId":"aa-briefcase-rating:anthropic-fable-5-1-card:QXJ0aWZpY2lhbCBBbmFseXNpczsgbWF4IGVmZm9ydDsgY29tcG9uZW50IHByZXNlbnRhdGlvbi4","id":"launch-anthropic-fable-5-1-card-84","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"human_pref","configuration":"Artificial Analysis; max effort; component presentation.","family":"AA-Briefcase","locator":"8.15.4","metric":"rating","notes":"","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":1572,"scoreUnit":"index","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"toolathlon-verified-june2026","evidenceKind":"lab_self_report","harnessId":"toolathlon-verified-june2026:anthropic-fable-5-1-card:UGFzc0AxOzEwOHRhc2tzL3RocmVlIHRyaWFsczsgaW50ZXJuYWwgaGFybmVzczsgbWF4IGVmZm9ydDsgRmFibGUgc2FmZWd1YXJkcytPcHVzNC44ZmFsbGJhY2s7IE9wdXMgc2FmZWd1YXJkcy9mYWxsYmFjayBkaXNhYmxlZDsgcGlubmVkIGNvbnRhaW5lcnMvZGF0YS4","id":"launch-anthropic-fable-5-1-card-85","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"agentic","configuration":"Pass@1;108tasks/three trials; internal harness; max effort; Fable safeguards+Opus4.8fallback; Opus safeguards/fallback disabled; pinned containers/data.","family":"Toolathlon","locator":"Table8.15.5.A","metric":"percent","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":77.8,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"toolathlon-verified-june2026","evidenceKind":"lab_self_report","harnessId":"toolathlon-verified-june2026:anthropic-fable-5-1-card:UGFzc0AxOzEwOHRhc2tzL3RocmVlIHRyaWFsczsgaW50ZXJuYWwgaGFybmVzczsgbWF4IGVmZm9ydDsgRmFibGUgc2FmZWd1YXJkcytPcHVzNC44ZmFsbGJhY2s7IE9wdXMgc2FmZWd1YXJkcy9mYWxsYmFjayBkaXNhYmxlZDsgcGlubmVkIGNvbnRhaW5lcnMvZGF0YS4","id":"launch-anthropic-fable-5-1-card-86","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"agentic","configuration":"Pass@1;108tasks/three trials; internal harness; max effort; Fable safeguards+Opus4.8fallback; Opus safeguards/fallback disabled; pinned containers/data.","family":"Toolathlon","locator":"Table8.15.5.A","metric":"percent","notes":"","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":80.6,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"toolathlon-verified-june2026","evidenceKind":"lab_self_report","harnessId":"toolathlon-verified-june2026:anthropic-fable-5-1-card:UGFzc0AzOzEwOHRhc2tzL3RocmVlIHRyaWFsczsgaW50ZXJuYWwgaGFybmVzczsgbWF4IGVmZm9ydDsgRmFibGUgc2FmZWd1YXJkcytPcHVzNC44ZmFsbGJhY2s7IE9wdXMgc2FmZWd1YXJkcy9mYWxsYmFjayBkaXNhYmxlZDsgcGlubmVkIGNvbnRhaW5lcnMvZGF0YS4","id":"launch-anthropic-fable-5-1-card-87","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"agentic","configuration":"Pass@3;108tasks/three trials; internal harness; max effort; Fable safeguards+Opus4.8fallback; Opus safeguards/fallback disabled; pinned containers/data.","family":"Toolathlon","locator":"Table8.15.5.A","metric":"percent","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":81.5,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"toolathlon-verified-june2026","evidenceKind":"lab_self_report","harnessId":"toolathlon-verified-june2026:anthropic-fable-5-1-card:UGFzc0AzOzEwOHRhc2tzL3RocmVlIHRyaWFsczsgaW50ZXJuYWwgaGFybmVzczsgbWF4IGVmZm9ydDsgRmFibGUgc2FmZWd1YXJkcytPcHVzNC44ZmFsbGJhY2s7IE9wdXMgc2FmZWd1YXJkcy9mYWxsYmFjayBkaXNhYmxlZDsgcGlubmVkIGNvbnRhaW5lcnMvZGF0YS4","id":"launch-anthropic-fable-5-1-card-88","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"agentic","configuration":"Pass@3;108tasks/three trials; internal harness; max effort; Fable safeguards+Opus4.8fallback; Opus safeguards/fallback disabled; pinned containers/data.","family":"Toolathlon","locator":"Table8.15.5.A","metric":"percent","notes":"","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":87,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"toolathlon-verified-june2026","evidenceKind":"lab_self_report","harnessId":"toolathlon-verified-june2026:anthropic-fable-5-1-card:UGFzc8KzOzEwOHRhc2tzL3RocmVlIHRyaWFsczsgaW50ZXJuYWwgaGFybmVzczsgbWF4IGVmZm9ydDsgRmFibGUgc2FmZWd1YXJkcytPcHVzNC44ZmFsbGJhY2s7IE9wdXMgc2FmZWd1YXJkcy9mYWxsYmFjayBkaXNhYmxlZDsgcGlubmVkIGNvbnRhaW5lcnMvZGF0YS4","id":"launch-anthropic-fable-5-1-card-89","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"agentic","configuration":"Pass³;108tasks/three trials; internal harness; max effort; Fable safeguards+Opus4.8fallback; Opus safeguards/fallback disabled; pinned containers/data.","family":"Toolathlon","locator":"Table8.15.5.A","metric":"percent","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":73.1,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"swe-bench-multimodal","evidenceKind":"lab_self_report","harnessId":"swe-bench-multimodal:anthropic-fable-5-1-card:QW50aHJvcGljIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IGFkYXB0aXZlIHRoaW5raW5nIGF0IG1heCBlZmZvcnQsIGRlZmF1bHQgc2FtcGxpbmcsIGZpdmUtdHJpYWwgbWVhbiB1bmxlc3Mgc2VjdGlvbiBzcGVjaWZpZXMgb3RoZXJ3aXNlOyBjb250ZXh0IGF0IG1vc3QgMU0uIENvbXBhcmF0b3IgY29uZmlndXJhdGlvbiBmb2xsb3dzIGNpdGVkIHByaW9yIGNhcmQgb3IgYm9hcmQuIEFnZW50IGlkZW50aXR5IGlzIG5vdCBlc3RhYmxpc2hlZCBhcyBtaW5pLXN3ZS1hZ2VudC4","id":"launch-anthropic-fable-5-1-card-9","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"coding","configuration":"Anthropic reported configuration; adaptive thinking at max effort, default sampling, five-trial mean unless section specifies otherwise; context at most 1M. Comparator configuration follows cited prior card or board. Agent identity is not established as mini-swe-agent.","family":"SWE-bench Multimodal","locator":"Table 8.1.A; section 8.2","metric":"percent","notes":"","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":59.4,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"toolathlon-verified-june2026","evidenceKind":"lab_self_report","harnessId":"toolathlon-verified-june2026:anthropic-fable-5-1-card:UGFzc8KzOzEwOHRhc2tzL3RocmVlIHRyaWFsczsgaW50ZXJuYWwgaGFybmVzczsgbWF4IGVmZm9ydDsgRmFibGUgc2FmZWd1YXJkcytPcHVzNC44ZmFsbGJhY2s7IE9wdXMgc2FmZWd1YXJkcy9mYWxsYmFjayBkaXNhYmxlZDsgcGlubmVkIGNvbnRhaW5lcnMvZGF0YS4","id":"launch-anthropic-fable-5-1-card-90","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"agentic","configuration":"Pass³;108tasks/three trials; internal harness; max effort; Fable safeguards+Opus4.8fallback; Opus safeguards/fallback disabled; pinned containers/data.","family":"Toolathlon","locator":"Table8.15.5.A","metric":"percent","notes":"","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":73.1,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"toolathlon-verified-june2026-turns","evidenceKind":"lab_self_report","harnessId":"toolathlon-verified-june2026-turns:anthropic-fable-5-1-card:YXZlcmFnZSB0dXJuczsxMDh0YXNrcy90aHJlZSB0cmlhbHM7IGludGVybmFsIGhhcm5lc3M7IG1heCBlZmZvcnQ7IEZhYmxlIHNhZmVndWFyZHMrT3B1czQuOGZhbGxiYWNrOyBPcHVzIHNhZmVndWFyZHMvZmFsbGJhY2sgZGlzYWJsZWQ7IHBpbm5lZCBjb250YWluZXJzL2RhdGEu","id":"launch-anthropic-fable-5-1-card-91","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"agentic","configuration":"average turns;108tasks/three trials; internal harness; max effort; Fable safeguards+Opus4.8fallback; Opus safeguards/fallback disabled; pinned containers/data.","family":"Toolathlon","locator":"Table8.15.5.A","metric":"turns","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":23.7,"scoreUnit":"index","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"toolathlon-verified-june2026-turns","evidenceKind":"lab_self_report","harnessId":"toolathlon-verified-june2026-turns:anthropic-fable-5-1-card:YXZlcmFnZSB0dXJuczsxMDh0YXNrcy90aHJlZSB0cmlhbHM7IGludGVybmFsIGhhcm5lc3M7IG1heCBlZmZvcnQ7IEZhYmxlIHNhZmVndWFyZHMrT3B1czQuOGZhbGxiYWNrOyBPcHVzIHNhZmVndWFyZHMvZmFsbGJhY2sgZGlzYWJsZWQ7IHBpbm5lZCBjb250YWluZXJzL2RhdGEu","id":"launch-anthropic-fable-5-1-card-92","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"agentic","configuration":"average turns;108tasks/three trials; internal harness; max effort; Fable safeguards+Opus4.8fallback; Opus safeguards/fallback disabled; pinned containers/data.","family":"Toolathlon","locator":"Table8.15.5.A","metric":"turns","notes":"","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":23.5,"scoreUnit":"index","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"automationbench","evidenceKind":"lab_self_report","harnessId":"automationbench:anthropic-fable-5-1-card:UHJpdmF0ZSBoZWxkLW91dCBib2FyZDsgc2ltdWxhdGVkIGJ1c2luZXNzLXdvcmtmbG93IGFwcCBBUElzOyBhbGwgYXNzZXJ0aW9ucyBtdXN0IHBhc3M7IG1heCBlZmZvcnQgc3RhdGVkIGZvciBGYWJsZTUuMS9PcHVzNS4","id":"launch-anthropic-fable-5-1-card-93","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"agentic","configuration":"Private held-out board; simulated business-workflow app APIs; all assertions must pass; max effort stated for Fable5.1/Opus5.","family":"AutomationBench","locator":"Table8.1.A;8.15.6","metric":"percent","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":31.4,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"automationbench","evidenceKind":"lab_self_report","harnessId":"automationbench:anthropic-fable-5-1-card:UHJpdmF0ZSBoZWxkLW91dCBib2FyZDsgc2ltdWxhdGVkIGJ1c2luZXNzLXdvcmtmbG93IGFwcCBBUElzOyBhbGwgYXNzZXJ0aW9ucyBtdXN0IHBhc3M7IG1heCBlZmZvcnQgc3RhdGVkIGZvciBGYWJsZTUuMS9PcHVzNS4","id":"launch-anthropic-fable-5-1-card-94","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"agentic","configuration":"Private held-out board; simulated business-workflow app APIs; all assertions must pass; max effort stated for Fable5.1/Opus5.","family":"AutomationBench","locator":"Table8.1.A;8.15.6","metric":"percent","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":17.05,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"automationbench","evidenceKind":"lab_self_report","harnessId":"automationbench:anthropic-fable-5-1-card:UHJpdmF0ZSBoZWxkLW91dCBib2FyZDsgc2ltdWxhdGVkIGJ1c2luZXNzLXdvcmtmbG93IGFwcCBBUElzOyBhbGwgYXNzZXJ0aW9ucyBtdXN0IHBhc3M7IG1heCBlZmZvcnQgc3RhdGVkIGZvciBGYWJsZTUuMS9PcHVzNS4","id":"launch-anthropic-fable-5-1-card-95","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"agentic","configuration":"Private held-out board; simulated business-workflow app APIs; all assertions must pass; max effort stated for Fable5.1/Opus5.","family":"AutomationBench","locator":"Table8.1.A;8.15.6","metric":"percent","notes":"","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":26.9,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"automationbench","evidenceKind":"lab_self_report","harnessId":"automationbench:anthropic-fable-5-1-card:UHJpdmF0ZSBoZWxkLW91dCBib2FyZDsgc2ltdWxhdGVkIGJ1c2luZXNzLXdvcmtmbG93IGFwcCBBUElzOyBhbGwgYXNzZXJ0aW9ucyBtdXN0IHBhc3M7IG1heCBlZmZvcnQgc3RhdGVkIGZvciBGYWJsZTUuMS9PcHVzNS4","id":"launch-anthropic-fable-5-1-card-96","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"agentic","configuration":"Private held-out board; simulated business-workflow app APIs; all assertions must pass; max effort stated for Fable5.1/Opus5.","family":"AutomationBench","locator":"Table8.1.A;8.15.6","metric":"percent","notes":"","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":19.6,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"arc-agi-1-percent","evidenceKind":"lab_self_report","harnessId":"arc-agi-1-percent:anthropic-fable-5-1-card:QVJDIFByaXplIHNlbWktcHJpdmF0ZSB2YWxpZGF0aW9uOyB2ZXJpZmllZCBGYWJsZTUuMSBtYXggZWZmb3J0OyBjb21wYXJhdG9yIGZpZ3VyZXMgYXMgcmVwb3J0ZWQgaW4gc3VtbWFyeS4","id":"launch-anthropic-fable-5-1-card-97","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"hard_reasoning","configuration":"ARC Prize semi-private validation; verified Fable5.1 max effort; comparator figures as reported in summary.","family":"ARC-AGI","locator":"Table8.1.A;8.16","metric":"percent","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":97.5,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"arc-agi-1-percent","evidenceKind":"lab_self_report","harnessId":"arc-agi-1-percent:anthropic-fable-5-1-card:QVJDIFByaXplIHNlbWktcHJpdmF0ZSB2YWxpZGF0aW9uOyB2ZXJpZmllZCBGYWJsZTUuMSBtYXggZWZmb3J0OyBjb21wYXJhdG9yIGZpZ3VyZXMgYXMgcmVwb3J0ZWQgaW4gc3VtbWFyeS4","id":"launch-anthropic-fable-5-1-card-98","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"hard_reasoning","configuration":"ARC Prize semi-private validation; verified Fable5.1 max effort; comparator figures as reported in summary.","family":"ARC-AGI","locator":"Table8.1.A;8.16","metric":"percent","notes":" Fable is the safeguarded deployed configuration; selected tasks may use disclosed Opus fallback, except evaluations explicitly counting safety blocks as failures.","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":98.5,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"arc-agi-1-percent","evidenceKind":"lab_self_report","harnessId":"arc-agi-1-percent:anthropic-fable-5-1-card:QVJDIFByaXplIHNlbWktcHJpdmF0ZSB2YWxpZGF0aW9uOyB2ZXJpZmllZCBGYWJsZTUuMSBtYXggZWZmb3J0OyBjb21wYXJhdG9yIGZpZ3VyZXMgYXMgcmVwb3J0ZWQgaW4gc3VtbWFyeS4","id":"launch-anthropic-fable-5-1-card-99","ingestRunId":"launch-anthropic-fable-5-1-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"b0d59edc7a60eef32a879c13d713cce60c3fefd7e6b5183afdc8b835af3c8c39","report":{"category":"hard_reasoning","configuration":"ARC Prize semi-private validation; verified Fable5.1 max effort; comparator figures as reported in summary.","family":"ARC-AGI","locator":"Table8.1.A;8.16","metric":"percent","notes":"","sourceId":"anthropic-fable-5-1-card","sourceTitle":"Claude Fable 5.1 and Claude Mythos 5.1 System Card"},"score":97.5,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"swe-bench-pro-percent-higher","evidenceKind":"lab_self_report","harnessId":"swe-bench-pro-percent-higher:anthropic-opus-5-5-card:QWRhcHRpdmUgdGhpbmtpbmcgYXQgbWF4IGVmZm9ydDsgZGVmYXVsdCBzYW1wbGluZzsgZml2ZS10cmlhbCBtZWFuOyBjb250ZXh0IGF0IG1vc3QgMU0uIFNXRS1iZW5jaCBQcm8gcHJvYmxlbXMgZnJvbSBhY3RpdmVseSBtYWludGFpbmVkIHJlcG9zaXRvcmllcyB3aXRoIGxhcmdlIG11bHRpLWZpbGUgZGlmZnMu","id":"launch-anthropic-opus-5-5-card-2262","ingestRunId":"launch-anthropic-opus-5-5-card","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-22","rawPayloadHash":"7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378","report":{"category":"coding","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Adaptive thinking at max effort; default sampling; five-trial mean; context at most 1M. SWE-bench Pro problems from actively maintained repositories with large multi-file diffs.","family":"SWE-bench Pro","locator":"Table 8.1.A; section 8.2","metric":"percent","notes":"Provider-published lab self-report from the Claude Opus 5.5 system card (SHA256 7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378). Not an independent board. Launch grid shows 89.9%.","sourceId":"anthropic-opus-5-5-card","sourceTitle":"Claude Opus 5.5 System Card"},"score":89.9,"scoreUnit":"percent","sourceUrl":"https://www-cdn.anthropic.com/fc1b44717c85dc068bc6ba5024219938094694bd/Claude%20Opus%205.5%20System%20Card.pdf"},{"benchmarkId":"swe-bench-multilingual-percent","evidenceKind":"lab_self_report","harnessId":"swe-bench-multilingual-percent:anthropic-opus-5-5-card:QWRhcHRpdmUgdGhpbmtpbmcgYXQgbWF4IGVmZm9ydDsgZGVmYXVsdCBzYW1wbGluZzsgZml2ZS10cmlhbCBtZWFuOyBjb250ZXh0IGF0IG1vc3QgMU0uIDMwMCBwcm9ibGVtcyBhY3Jvc3MgbmluZSBwcm9ncmFtbWluZyBsYW5ndWFnZXMu","id":"launch-anthropic-opus-5-5-card-2263","ingestRunId":"launch-anthropic-opus-5-5-card","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-22","rawPayloadHash":"7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378","report":{"category":"coding","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Adaptive thinking at max effort; default sampling; five-trial mean; context at most 1M. 300 problems across nine programming languages.","family":"SWE-bench Multilingual","locator":"Table 8.1.A; section 8.2","metric":"percent","notes":"Provider-published lab self-report from the Claude Opus 5.5 system card (SHA256 7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378). Not an independent board. Launch grid shows 93.9%.","sourceId":"anthropic-opus-5-5-card","sourceTitle":"Claude Opus 5.5 System Card"},"score":93.9,"scoreUnit":"percent","sourceUrl":"https://www-cdn.anthropic.com/fc1b44717c85dc068bc6ba5024219938094694bd/Claude%20Opus%205.5%20System%20Card.pdf"},{"benchmarkId":"swe-bench-multimodal","evidenceKind":"lab_self_report","harnessId":"swe-bench-multimodal:anthropic-opus-5-5-card:QWRhcHRpdmUgdGhpbmtpbmcgYXQgbWF4IGVmZm9ydDsgZGVmYXVsdCBzYW1wbGluZzsgZml2ZS10cmlhbCBtZWFuOyBjb250ZXh0IGF0IG1vc3QgMU0uIFZpc3VhbCBjb250ZXh0IGFkZGVkIHRvIGlzc3VlIGRlc2NyaXB0aW9ucy4","id":"launch-anthropic-opus-5-5-card-2264","ingestRunId":"launch-anthropic-opus-5-5-card","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-22","rawPayloadHash":"7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378","report":{"category":"coding","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Adaptive thinking at max effort; default sampling; five-trial mean; context at most 1M. Visual context added to issue descriptions.","family":"SWE-bench Multimodal","locator":"Table 8.1.A; section 8.2","metric":"percent","notes":"Provider-published lab self-report from the Claude Opus 5.5 system card (SHA256 7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378). Not an independent board.","sourceId":"anthropic-opus-5-5-card","sourceTitle":"Claude Opus 5.5 System Card"},"score":61.4,"scoreUnit":"percent","sourceUrl":"https://www-cdn.anthropic.com/fc1b44717c85dc068bc6ba5024219938094694bd/Claude%20Opus%205.5%20System%20Card.pdf"},{"benchmarkId":"deepswe-1-1-percent","evidenceKind":"lab_self_report","harnessId":"deepswe-1-1-percent:anthropic-opus-5-5-card:MTEzIGxvbmctaG9yaXpvbiB0YXNrczsgZml2ZS10cmlhbCBtZWFuLiBTZWN0aW9uIDguMyBkb2VzIG5vdCBzdGF0ZSByZWFzb25pbmcgZWZmb3J0Lg","id":"launch-anthropic-opus-5-5-card-2265","ingestRunId":"launch-anthropic-opus-5-5-card","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-22","rawPayloadHash":"7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378","report":{"category":"coding","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"113 long-horizon tasks; five-trial mean. Section 8.3 does not state reasoning effort.","family":"DeepSWE","locator":"section 8.3","metric":"percent","notes":"Provider-published lab self-report from the Claude Opus 5.5 system card (SHA256 7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378). Not an independent board. Effort is not stated in this section, so it is not copied from the Table 8.1.A max-effort default.","sourceId":"anthropic-opus-5-5-card","sourceTitle":"Claude Opus 5.5 System Card"},"score":74.2,"scoreUnit":"percent","sourceUrl":"https://www-cdn.anthropic.com/fc1b44717c85dc068bc6ba5024219938094694bd/Claude%20Opus%205.5%20System%20Card.pdf"},{"benchmarkId":"frontiercode-1-1-main","evidenceKind":"lab_self_report","harnessId":"frontiercode-1-1-main:anthropic-opus-5-5-card:Q29nbml0aW9uIGFnZW50aWMgY29kaW5nIGluIENsYXVkZSBDb2RlOyBjb21wb3NpdGUgZnVuY3Rpb25hbCBhbmQgY29kZS1xdWFsaXR5IHNjb3JlOyBtYXggZWZmb3J0OyBtZWFuQDUuIENvZ25pdGlvbiByYW4gdGhlIGV2YWx1YXRpb24u","id":"launch-anthropic-opus-5-5-card-2266","ingestRunId":"launch-anthropic-opus-5-5-card","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-22","rawPayloadHash":"7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378","report":{"category":"coding","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Cognition agentic coding in Claude Code; composite functional and code-quality score; max effort; mean@5. Cognition ran the evaluation.","family":"FrontierCode","locator":"Table 8.1.A; section 8.4","metric":"percent","notes":"Provider-published lab self-report from the Claude Opus 5.5 system card (SHA256 7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378). Not an independent board. Launch grid shows 54.4% for FrontierCode v1.1 (Main). Section 8.4 says performance falls above medium effort and this max-effort score is 54.4%.","sourceId":"anthropic-opus-5-5-card","sourceTitle":"Claude Opus 5.5 System Card"},"score":54.4,"scoreUnit":"percent","sourceUrl":"https://www-cdn.anthropic.com/fc1b44717c85dc068bc6ba5024219938094694bd/Claude%20Opus%205.5%20System%20Card.pdf"},{"benchmarkId":"frontiercode-1-1-main","evidenceKind":"lab_self_report","harnessId":"frontiercode-1-1-main:anthropic-opus-5-5-card:Q29nbml0aW9uIGFnZW50aWMgY29kaW5nIGluIENsYXVkZSBDb2RlOyBjb21wb3NpdGUgZnVuY3Rpb25hbCBhbmQgY29kZS1xdWFsaXR5IHNjb3JlOyBtZWRpdW0gZWZmb3J0OyBtZWFuQDUuIEhpZ2hlc3QgTWFpbiBzY29yZS4gQ29nbml0aW9uIHJhbiB0aGUgZXZhbHVhdGlvbi4","id":"launch-anthropic-opus-5-5-card-2267","ingestRunId":"launch-anthropic-opus-5-5-card","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-22","rawPayloadHash":"7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378","report":{"category":"coding","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Cognition agentic coding in Claude Code; composite functional and code-quality score; medium effort; mean@5. Highest Main score. Cognition ran the evaluation.","family":"FrontierCode","locator":"section 8.4","metric":"percent","notes":"Provider-published lab self-report from the Claude Opus 5.5 system card (SHA256 7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378). Not an independent board. Launch-page prose also states 54.6% at default (medium) effort.","sourceId":"anthropic-opus-5-5-card","sourceTitle":"Claude Opus 5.5 System Card"},"score":54.6,"scoreUnit":"percent","sourceUrl":"https://www-cdn.anthropic.com/fc1b44717c85dc068bc6ba5024219938094694bd/Claude%20Opus%205.5%20System%20Card.pdf"},{"benchmarkId":"frontiercode-1-1-extended","evidenceKind":"lab_self_report","harnessId":"frontiercode-1-1-extended:anthropic-opus-5-5-card:Q29nbml0aW9uIGFnZW50aWMgY29kaW5nIGluIENsYXVkZSBDb2RlOyBjb21wb3NpdGUgZnVuY3Rpb25hbCBhbmQgY29kZS1xdWFsaXR5IHNjb3JlOyBtZWRpdW0gZWZmb3J0OyBtZWFuQDUuIEhpZ2hlc3QgRXh0ZW5kZWQgc2NvcmUuIENvZ25pdGlvbiByYW4gdGhlIGV2YWx1YXRpb24u","id":"launch-anthropic-opus-5-5-card-2268","ingestRunId":"launch-anthropic-opus-5-5-card","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-22","rawPayloadHash":"7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378","report":{"category":"coding","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Cognition agentic coding in Claude Code; composite functional and code-quality score; medium effort; mean@5. Highest Extended score. Cognition ran the evaluation.","family":"FrontierCode","locator":"section 8.4","metric":"percent","notes":"Provider-published lab self-report from the Claude Opus 5.5 system card (SHA256 7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378). Not an independent board.","sourceId":"anthropic-opus-5-5-card","sourceTitle":"Claude Opus 5.5 System Card"},"score":65.3,"scoreUnit":"percent","sourceUrl":"https://www-cdn.anthropic.com/fc1b44717c85dc068bc6ba5024219938094694bd/Claude%20Opus%205.5%20System%20Card.pdf"},{"benchmarkId":"frontiercode-1-1-extended","evidenceKind":"lab_self_report","harnessId":"frontiercode-1-1-extended:anthropic-opus-5-5-card:Q29nbml0aW9uIGFnZW50aWMgY29kaW5nIGluIENsYXVkZSBDb2RlOyBjb21wb3NpdGUgZnVuY3Rpb25hbCBhbmQgY29kZS1xdWFsaXR5IHNjb3JlOyBtYXggZWZmb3J0OyBtZWFuQDUuIENvZ25pdGlvbiByYW4gdGhlIGV2YWx1YXRpb24u","id":"launch-anthropic-opus-5-5-card-2269","ingestRunId":"launch-anthropic-opus-5-5-card","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-22","rawPayloadHash":"7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378","report":{"category":"coding","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Cognition agentic coding in Claude Code; composite functional and code-quality score; max effort; mean@5. Cognition ran the evaluation.","family":"FrontierCode","locator":"section 8.4","metric":"percent","notes":"Provider-published lab self-report from the Claude Opus 5.5 system card (SHA256 7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378). Not an independent board.","sourceId":"anthropic-opus-5-5-card","sourceTitle":"Claude Opus 5.5 System Card"},"score":63.6,"scoreUnit":"percent","sourceUrl":"https://www-cdn.anthropic.com/fc1b44717c85dc068bc6ba5024219938094694bd/Claude%20Opus%205.5%20System%20Card.pdf"},{"benchmarkId":"terminal-bench-4-0-percent","evidenceKind":"lab_self_report","harnessId":"terminal-bench-4-0-percent:anthropic-opus-5-5-card:Q2xhdWRlIENvZGUgLS1iYXJlOyB4aGlnaCB0aGlua2luZyBlZmZvcnQ7IHNhZmVndWFyZHMgZW5hYmxlZCB3aXRoIHNlcnZlci1zaWRlIGZhbGxiYWNrICgyLjUlIG9mIHJlcXVlc3RzLCAxMCUgb2YgdHJpYWxzKTsgZml2ZSB0cmlhbHMgcGVyIHRhc2sgKDMzMCB0cmlhbHMpIG9uIDY2IHRhc2tzLg","id":"launch-anthropic-opus-5-5-card-2270","ingestRunId":"launch-anthropic-opus-5-5-card","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-22","rawPayloadHash":"7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378","report":{"category":"agentic","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Claude Code --bare; xhigh thinking effort; safeguards enabled with server-side fallback (2.5% of requests, 10% of trials); five trials per task (330 trials) on 66 tasks.","family":"Terminal-Bench","locator":"section 8.5","metric":"percent","notes":"Provider-published lab self-report from the Claude Opus 5.5 system card (SHA256 7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378). Not an independent board. Table 8.1.A and the launch grid round this xhigh result to 66.4%. Section 8.5 states 66.36%.","sourceId":"anthropic-opus-5-5-card","sourceTitle":"Claude Opus 5.5 System Card"},"score":66.36,"scoreUnit":"percent","sourceUrl":"https://www-cdn.anthropic.com/fc1b44717c85dc068bc6ba5024219938094694bd/Claude%20Opus%205.5%20System%20Card.pdf"},{"benchmarkId":"terminal-bench-4-0-percent","evidenceKind":"lab_self_report","harnessId":"terminal-bench-4-0-percent:anthropic-opus-5-5-card:Q2xhdWRlIENvZGUgLS1iYXJlOyBtYXggdGhpbmtpbmcgZWZmb3J0OyBzYWZlZ3VhcmRzIGVuYWJsZWQgd2l0aCB0aGUgZGVmYXVsdCBzZXJ2ZXItc2lkZSBmYWxsYmFjazsgZml2ZSB0cmlhbHMgcGVyIHRhc2sgb24gNjYgdGFza3MuIFNlY3Rpb24gOC41IHNheXMgdGhpcyBpcyB3aXRoaW4gbm9pc2Ugb2YgeGhpZ2gu","id":"launch-anthropic-opus-5-5-card-2271","ingestRunId":"launch-anthropic-opus-5-5-card","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-22","rawPayloadHash":"7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378","report":{"category":"agentic","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Claude Code --bare; max thinking effort; safeguards enabled with the default server-side fallback; five trials per task on 66 tasks. Section 8.5 says this is within noise of xhigh.","family":"Terminal-Bench","locator":"section 8.5","metric":"percent","notes":"Provider-published lab self-report from the Claude Opus 5.5 system card (SHA256 7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378). Not an independent board.","sourceId":"anthropic-opus-5-5-card","sourceTitle":"Claude Opus 5.5 System Card"},"score":64.8,"scoreUnit":"percent","sourceUrl":"https://www-cdn.anthropic.com/fc1b44717c85dc068bc6ba5024219938094694bd/Claude%20Opus%205.5%20System%20Card.pdf"},{"benchmarkId":"terminal-bench-science-0-1-percent","evidenceKind":"lab_self_report","harnessId":"terminal-bench-science-0-1-percent:anthropic-opus-5-5-card:Q2xhdWRlIENvZGUgLS1iYXJlOyBtYXggdGhpbmtpbmcgZWZmb3J0OyBzYWZlZ3VhcmRzIGVuYWJsZWQgd2l0aCBzZXJ2ZXItc2lkZSBmYWxsYmFjayAoMy45JSBvZiByZXF1ZXN0cywgNSUgb2YgdHJpYWxzKTsgMTAgdHJpYWxzIHBlciB0YXNrICg3MDAgdHJpYWxzKSBvbiA3MCB0YXNrcy4","id":"launch-anthropic-opus-5-5-card-2272","ingestRunId":"launch-anthropic-opus-5-5-card","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-22","rawPayloadHash":"7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378","report":{"category":"agentic","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Claude Code --bare; max thinking effort; safeguards enabled with server-side fallback (3.9% of requests, 5% of trials); 10 trials per task (700 trials) on 70 tasks.","family":"Terminal-Bench","locator":"Table 8.1.A; section 8.6","metric":"percent","notes":"Provider-published lab self-report from the Claude Opus 5.5 system card (SHA256 7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378). Not an independent board. The Anthropic launch grid at https://www.anthropic.com/claude-opus-5-5 shows 58.7%.","sourceId":"anthropic-opus-5-5-card","sourceTitle":"Claude Opus 5.5 System Card"},"score":58.7,"scoreUnit":"percent","sourceUrl":"https://www-cdn.anthropic.com/fc1b44717c85dc068bc6ba5024219938094694bd/Claude%20Opus%205.5%20System%20Card.pdf"},{"benchmarkId":"frontierswe-2-percent","evidenceKind":"lab_self_report","harnessId":"frontierswe-2-percent:anthropic-opus-5-5-card:UHJveGltYWwgYWdlbnQgaGFybmVzczsgbWF4IHJlYXNvbmluZyBlZmZvcnQ7IDM0IHRhc2tzOyBmaXZlIHRyaWFscyBwZXIgdGFzazsgbWVhbiBhY3Jvc3MgdHJpYWxzLg","id":"launch-anthropic-opus-5-5-card-2273","ingestRunId":"launch-anthropic-opus-5-5-card","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-22","rawPayloadHash":"7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378","report":{"category":"coding","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Proximal agent harness; max reasoning effort; 34 tasks; five trials per task; mean across trials.","family":"FrontierSWE","locator":"section 8.7","metric":"percent","notes":"Provider-published lab self-report from the Claude Opus 5.5 system card (SHA256 7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378). Not an independent board.","sourceId":"anthropic-opus-5-5-card","sourceTitle":"Claude Opus 5.5 System Card"},"score":62.3,"scoreUnit":"percent","sourceUrl":"https://www-cdn.anthropic.com/fc1b44717c85dc068bc6ba5024219938094694bd/Claude%20Opus%205.5%20System%20Card.pdf"},{"benchmarkId":"cursorbench-4-0-percent","evidenceKind":"lab_self_report","harnessId":"cursorbench-4-0-percent:anthropic-opus-5-5-card:Q3Vyc29yIHByb2R1Y3Rpb24gYWdlbnQgaGFybmVzczsgbWF4IGVmZm9ydC4gSW5kZXBlbmRlbnRseSBtZWFzdXJlZCBieSBDdXJzb3I7IEFudGhyb3BpYyBlc3RpbWF0ZWQgY29zdCBmcm9tIEN1cnNvciB0b2tlbiBjb3VudHMu","id":"launch-anthropic-opus-5-5-card-2274","ingestRunId":"launch-anthropic-opus-5-5-card","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-22","rawPayloadHash":"7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378","report":{"category":"coding","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Cursor production agent harness; max effort. Independently measured by Cursor; Anthropic estimated cost from Cursor token counts.","family":"CursorBench","locator":"section 8.8","metric":"percent","notes":"Provider-published lab self-report from the Claude Opus 5.5 system card (SHA256 7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378). Not an independent board. Launch grid shows 57.8%.","sourceId":"anthropic-opus-5-5-card","sourceTitle":"Claude Opus 5.5 System Card"},"score":57.8,"scoreUnit":"percent","sourceUrl":"https://www-cdn.anthropic.com/fc1b44717c85dc068bc6ba5024219938094694bd/Claude%20Opus%205.5%20System%20Card.pdf"},{"benchmarkId":"cursorbench-4-0-percent","evidenceKind":"lab_self_report","harnessId":"cursorbench-4-0-percent:anthropic-opus-5-5-card:Q3Vyc29yIHByb2R1Y3Rpb24gYWdlbnQgaGFybmVzczsgeGhpZ2ggZWZmb3J0LiBJbmRlcGVuZGVudGx5IG1lYXN1cmVkIGJ5IEN1cnNvci4","id":"launch-anthropic-opus-5-5-card-2275","ingestRunId":"launch-anthropic-opus-5-5-card","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-22","rawPayloadHash":"7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378","report":{"category":"coding","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Cursor production agent harness; xhigh effort. Independently measured by Cursor.","family":"CursorBench","locator":"section 8.8","metric":"percent","notes":"Provider-published lab self-report from the Claude Opus 5.5 system card (SHA256 7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378). Not an independent board.","sourceId":"anthropic-opus-5-5-card","sourceTitle":"Claude Opus 5.5 System Card"},"score":56,"scoreUnit":"percent","sourceUrl":"https://www-cdn.anthropic.com/fc1b44717c85dc068bc6ba5024219938094694bd/Claude%20Opus%205.5%20System%20Card.pdf"},{"benchmarkId":"cursorbench-4-0-percent","evidenceKind":"lab_self_report","harnessId":"cursorbench-4-0-percent:anthropic-opus-5-5-card:Q3Vyc29yIHByb2R1Y3Rpb24gYWdlbnQgaGFybmVzczsgaGlnaCBlZmZvcnQuIEluZGVwZW5kZW50bHkgbWVhc3VyZWQgYnkgQ3Vyc29yLg","id":"launch-anthropic-opus-5-5-card-2276","ingestRunId":"launch-anthropic-opus-5-5-card","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-22","rawPayloadHash":"7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378","report":{"category":"coding","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Cursor production agent harness; high effort. Independently measured by Cursor.","family":"CursorBench","locator":"section 8.8","metric":"percent","notes":"Provider-published lab self-report from the Claude Opus 5.5 system card (SHA256 7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378). Not an independent board.","sourceId":"anthropic-opus-5-5-card","sourceTitle":"Claude Opus 5.5 System Card"},"score":56,"scoreUnit":"percent","sourceUrl":"https://www-cdn.anthropic.com/fc1b44717c85dc068bc6ba5024219938094694bd/Claude%20Opus%205.5%20System%20Card.pdf"},{"benchmarkId":"cursorbench-4-0-percent","evidenceKind":"lab_self_report","harnessId":"cursorbench-4-0-percent:anthropic-opus-5-5-card:Q3Vyc29yIHByb2R1Y3Rpb24gYWdlbnQgaGFybmVzczsgbWVkaXVtIGVmZm9ydC4gSW5kZXBlbmRlbnRseSBtZWFzdXJlZCBieSBDdXJzb3Iu","id":"launch-anthropic-opus-5-5-card-2277","ingestRunId":"launch-anthropic-opus-5-5-card","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-22","rawPayloadHash":"7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378","report":{"category":"coding","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Cursor production agent harness; medium effort. Independently measured by Cursor.","family":"CursorBench","locator":"section 8.8","metric":"percent","notes":"Provider-published lab self-report from the Claude Opus 5.5 system card (SHA256 7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378). Not an independent board. Launch-page prose also states 52.5% at default (medium) effort.","sourceId":"anthropic-opus-5-5-card","sourceTitle":"Claude Opus 5.5 System Card"},"score":52.5,"scoreUnit":"percent","sourceUrl":"https://www-cdn.anthropic.com/fc1b44717c85dc068bc6ba5024219938094694bd/Claude%20Opus%205.5%20System%20Card.pdf"},{"benchmarkId":"programbench-166-golden-task-subset","evidenceKind":"lab_self_report","harnessId":"programbench-166-golden-task-subset:anthropic-opus-5-5-card:bWluaS1zd2UtYWdlbnQgd2l0aG91dCB0aGUgc2l4LWhvdXIgdGltZW91dDsgMzQgdGFza3Mgd2l0aCBhIHJlZmVyZW5jZSBiaW5hcnkgYmVsb3cgMC45IGV4Y2x1ZGVkOyBzY29yZWQgb25seSBvbiB0ZXN0cyB0aGUgcmVmZXJlbmNlIGJpbmFyeSBwYXNzZXM7IGNvbnRleHQgdXAgdG8gMU0uIFNlY3Rpb24gOC4xMC4xIGRvZXMgbm90IHN0YXRlIHJlYXNvbmluZyBlZmZvcnQu","id":"launch-anthropic-opus-5-5-card-2278","ingestRunId":"launch-anthropic-opus-5-5-card","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-22","rawPayloadHash":"7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378","report":{"category":"long_context","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"mini-swe-agent without the six-hour timeout; 34 tasks with a reference binary below 0.9 excluded; scored only on tests the reference binary passes; context up to 1M. Section 8.10.1 does not state reasoning effort.","family":"ProgramBench","locator":"section 8.10.1","metric":"percent","notes":"Provider-published lab self-report from the Claude Opus 5.5 system card (SHA256 7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378). Not an independent board. Hidden-test pass rate, not the fraction of completely solved programs. Effort is not stated in this section.","sourceId":"anthropic-opus-5-5-card","sourceTitle":"Claude Opus 5.5 System Card"},"score":91.2,"scoreUnit":"percent","sourceUrl":"https://www-cdn.anthropic.com/fc1b44717c85dc068bc6ba5024219938094694bd/Claude%20Opus%205.5%20System%20Card.pdf"},{"benchmarkId":"humanity-s-last-exam","evidenceKind":"lab_self_report","harnessId":"humanity-s-last-exam:anthropic-opus-5-5-card:RnVsbCAyLDUwMCBxdWVzdGlvbnM7IG5vIHRvb2xzOyBzZWN0aW9uIDguMTEuMSBzZXRzIHRoaW5raW5nIHRvIGF1dG8sIGEgMU0gdG90YWwgdG9rZW4gY2FwLCBubyBjb21wYWN0aW9uLCBhbmQgYW4gT3B1cyA0LjYgZ3JhZGVyLg","id":"launch-anthropic-opus-5-5-card-2279","ingestRunId":"launch-anthropic-opus-5-5-card","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-22","rawPayloadHash":"7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378","report":{"category":"hard_reasoning","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Full 2,500 questions; no tools; section 8.11.1 sets thinking to auto, a 1M total token cap, no compaction, and an Opus 4.6 grader.","family":"Humanity's Last Exam","locator":"Table 8.1.A; section 8.11.1","metric":"percent","notes":"Provider-published lab self-report from the Claude Opus 5.5 system card (SHA256 7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378). Not an independent board. Table 8.1.A's default note says adaptive max effort unless otherwise noted. Section 8.11.1 states thinking was set to auto for these runs.","sourceId":"anthropic-opus-5-5-card","sourceTitle":"Claude Opus 5.5 System Card"},"score":64.4,"scoreUnit":"percent","sourceUrl":"https://www-cdn.anthropic.com/fc1b44717c85dc068bc6ba5024219938094694bd/Claude%20Opus%205.5%20System%20Card.pdf"},{"benchmarkId":"humanity-s-last-exam","evidenceKind":"lab_self_report","harnessId":"humanity-s-last-exam:anthropic-opus-5-5-card:RnVsbCAyLDUwMCBxdWVzdGlvbnM7IHdlYiBzZWFyY2gsIHdlYiBmZXRjaCwgcHJvZ3JhbW1hdGljIHRvb2wgY2FsbGluZywgYW5kIGNvZGUgZXhlY3V0aW9uOyB0aGlua2luZyBzZXQgdG8gYXV0bzsgMU0gdG90YWwgdG9rZW4gY2FwOyBubyBjb21wYWN0aW9uOyBPcHVzIDQuNiBncmFkZXI7IEhMRSBzb3VyY2UgYmxvY2tsaXN0IGFuZCBjb250YW1pbmF0aW9uIHJldmlldy4","id":"launch-anthropic-opus-5-5-card-2280","ingestRunId":"launch-anthropic-opus-5-5-card","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-22","rawPayloadHash":"7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378","report":{"category":"hard_reasoning","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Full 2,500 questions; web search, web fetch, programmatic tool calling, and code execution; thinking set to auto; 1M total token cap; no compaction; Opus 4.6 grader; HLE source blocklist and contamination review.","family":"Humanity's Last Exam","locator":"Table 8.1.A; section 8.11.1","metric":"percent","notes":"Provider-published lab self-report from the Claude Opus 5.5 system card (SHA256 7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378). Not an independent board. Launch grid shows 67.7% with tools.","sourceId":"anthropic-opus-5-5-card","sourceTitle":"Claude Opus 5.5 System Card"},"score":67.7,"scoreUnit":"percent","sourceUrl":"https://www-cdn.anthropic.com/fc1b44717c85dc068bc6ba5024219938094694bd/Claude%20Opus%205.5%20System%20Card.pdf"},{"benchmarkId":"chartography","evidenceKind":"lab_self_report","harnessId":"chartography:anthropic-opus-5-5-card:MTAwIHRhc2tzOyBhZGFwdGl2ZSB0aGlua2luZyBhdCBtYXggZWZmb3J0OyBmaXZlIHJ1bnM7IG5vIHRvb2xzOyBHZW1pbmkgMy41IEZsYXNoIGdyYWRlcjsgZXhwZXJ0IGFjY2VwdGFibGUgcmFuZ2VzLg","id":"launch-anthropic-opus-5-5-card-2281","ingestRunId":"launch-anthropic-opus-5-5-card","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-22","rawPayloadHash":"7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378","report":{"category":"multimodal","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"100 tasks; adaptive thinking at max effort; five runs; no tools; Gemini 3.5 Flash grader; expert acceptable ranges.","family":"Chartography","locator":"section 8.13.1","metric":"percent","notes":"Provider-published lab self-report from the Claude Opus 5.5 system card (SHA256 7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378). Not an independent board.","sourceId":"anthropic-opus-5-5-card","sourceTitle":"Claude Opus 5.5 System Card"},"score":64.4,"scoreUnit":"percent","sourceUrl":"https://www-cdn.anthropic.com/fc1b44717c85dc068bc6ba5024219938094694bd/Claude%20Opus%205.5%20System%20Card.pdf"},{"benchmarkId":"chartography","evidenceKind":"lab_self_report","harnessId":"chartography:anthropic-opus-5-5-card:MTAwIHRhc2tzOyBhZGFwdGl2ZSB0aGlua2luZyBhdCBtYXggZWZmb3J0OyBmaXZlIHJ1bnM7IHdpdGggdG9vbHMgKGNvbnRhaW5lciwgaW1hZ2UgZmlsZSwgc3RhbmRhcmQgbGlicmFyaWVzLCBhbmQgYW4gaW1hZ2UgY3JvcHBpbmcgdG9vbCk7IEdlbWluaSAzLjUgRmxhc2ggZ3JhZGVyLg","id":"launch-anthropic-opus-5-5-card-2282","ingestRunId":"launch-anthropic-opus-5-5-card","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-22","rawPayloadHash":"7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378","report":{"category":"multimodal","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"100 tasks; adaptive thinking at max effort; five runs; with tools (container, image file, standard libraries, and an image cropping tool); Gemini 3.5 Flash grader.","family":"Chartography","locator":"section 8.13.1","metric":"percent","notes":"Provider-published lab self-report from the Claude Opus 5.5 system card (SHA256 7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378). Not an independent board. Launch grid shows 89.0% with tools.","sourceId":"anthropic-opus-5-5-card","sourceTitle":"Claude Opus 5.5 System Card"},"score":89,"scoreUnit":"percent","sourceUrl":"https://www-cdn.anthropic.com/fc1b44717c85dc068bc6ba5024219938094694bd/Claude%20Opus%205.5%20System%20Card.pdf"},{"benchmarkId":"benchcad-vision2code-1000-file-subset","evidenceKind":"lab_self_report","harnessId":"benchcad-vision2code-1000-file-subset:anthropic-opus-5-5-card:UmFuZG9tIDEsMDAwIG9mIDE3LDkwMCBWaXNpb24yQ29kZSBmaWxlczsgZml2ZSBydW5zOyBhZGFwdGl2ZSB0aGlua2luZyBhdCBtYXggZWZmb3J0OyBubyB0b29sczsgdmlld3MgcmVuZGVyZWQgYXQgMjU2eDI1NiBweCwgdGhlIHJlc29sdXRpb24gQW50aHJvcGljIHNheXMgbWF0Y2hlcyB0aGUgcmVmZXJlbmNlIGltcGxlbWVudGF0aW9uLg","id":"launch-anthropic-opus-5-5-card-2283","ingestRunId":"launch-anthropic-opus-5-5-card","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-22","rawPayloadHash":"7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378","report":{"category":"multimodal","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Random 1,000 of 17,900 Vision2Code files; five runs; adaptive thinking at max effort; no tools; views rendered at 256x256 px, the resolution Anthropic says matches the reference implementation.","family":"BenchCAD","locator":"section 8.13.2","metric":"voxel IoU","notes":"Provider-published lab self-report from the Claude Opus 5.5 system card (SHA256 7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378). Not an independent board. Earlier cards used 128x128 px renders. This section publishes the corrected resolution.","sourceId":"anthropic-opus-5-5-card","sourceTitle":"Claude Opus 5.5 System Card"},"score":0.73,"scoreUnit":"index","sourceUrl":"https://www-cdn.anthropic.com/fc1b44717c85dc068bc6ba5024219938094694bd/Claude%20Opus%205.5%20System%20Card.pdf"},{"benchmarkId":"benchcad-vision2code-1000-file-subset","evidenceKind":"lab_self_report","harnessId":"benchcad-vision2code-1000-file-subset:anthropic-opus-5-5-card:UmFuZG9tIDEsMDAwIG9mIDE3LDkwMCBWaXNpb24yQ29kZSBmaWxlczsgZml2ZSBydW5zOyBhZGFwdGl2ZSB0aGlua2luZyBhdCBtYXggZWZmb3J0OyB3aXRoIHRvb2xzIChjb250YWluZXIsIGltYWdlIGZpbGVzLCBzdGFuZGFyZCBsaWJyYXJpZXMsIGFuZCBhbiBpbWFnZSBjcm9wcGluZyB0b29sKTsgMjU2eDI1NiBweCB2aWV3cy4","id":"launch-anthropic-opus-5-5-card-2284","ingestRunId":"launch-anthropic-opus-5-5-card","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-22","rawPayloadHash":"7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378","report":{"category":"multimodal","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Random 1,000 of 17,900 Vision2Code files; five runs; adaptive thinking at max effort; with tools (container, image files, standard libraries, and an image cropping tool); 256x256 px views.","family":"BenchCAD","locator":"section 8.13.2","metric":"voxel IoU","notes":"Provider-published lab self-report from the Claude Opus 5.5 system card (SHA256 7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378). Not an independent board.","sourceId":"anthropic-opus-5-5-card","sourceTitle":"Claude Opus 5.5 System Card"},"score":0.962,"scoreUnit":"index","sourceUrl":"https://www-cdn.anthropic.com/fc1b44717c85dc068bc6ba5024219938094694bd/Claude%20Opus%205.5%20System%20Card.pdf"},{"benchmarkId":"osworld-2-0-september-10-2026-task-release","evidenceKind":"lab_self_report","harnessId":"osworld-2-0-september-10-2026-task-release:anthropic-opus-5-5-card:UGFydGlhbCBzY29yZSwgcGFzc0AxOyAxMDggdGFza3M7IGZpdmUgcnVuczsgMTA4MHA7IDUwMCBhY3Rpb24gc3RlcHM7IG1heCByZWFzb25pbmcgZWZmb3J0OyBPcHVzIDQuOCBncmFkZXIgd2hlcmUgcmVxdWlyZWQuIFNlcHRlbWJlciAxMCwgMjAyNiB0YXNrIGZpbGVzIGFuZCBzZXJ2ZXItc2lkZSBjb250ZXh0IGNvbXBhY3Rpb24gYWZ0ZXIgMTAwayB0b2tlbnMu","id":"launch-anthropic-opus-5-5-card-2285","ingestRunId":"launch-anthropic-opus-5-5-card","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-22","rawPayloadHash":"7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378","report":{"category":"agentic","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Partial score, pass@1; 108 tasks; five runs; 1080p; 500 action steps; max reasoning effort; Opus 4.8 grader where required. September 10, 2026 task files and server-side context compaction after 100k tokens.","family":"OSWorld","locator":"Table 8.1.A; section 8.13.3","metric":"percent","notes":"Provider-published lab self-report from the Claude Opus 5.5 system card (SHA256 7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378). Not an independent board. Launch grid shows 81.8% partial. Section 8.13.3 says these settings supersede the Fable 5.1 card's OSWorld 2.0 figures.","sourceId":"anthropic-opus-5-5-card","sourceTitle":"Claude Opus 5.5 System Card"},"score":81.8,"scoreUnit":"percent","sourceUrl":"https://www-cdn.anthropic.com/fc1b44717c85dc068bc6ba5024219938094694bd/Claude%20Opus%205.5%20System%20Card.pdf"},{"benchmarkId":"osworld-2-0-september-10-2026-task-release","evidenceKind":"lab_self_report","harnessId":"osworld-2-0-september-10-2026-task-release:anthropic-opus-5-5-card:U3RyaWN0IHBhc3MgcmF0ZSwgcGFzc0AxOyAxMDggdGFza3M7IGZpdmUgcnVuczsgMTA4MHA7IDUwMCBhY3Rpb24gc3RlcHM7IG1heCByZWFzb25pbmcgZWZmb3J0OyBPcHVzIDQuOCBncmFkZXIgd2hlcmUgcmVxdWlyZWQuIFNlcHRlbWJlciAxMCwgMjAyNiB0YXNrIGZpbGVzIGFuZCBzZXJ2ZXItc2lkZSBjb250ZXh0IGNvbXBhY3Rpb24gYWZ0ZXIgMTAwayB0b2tlbnMu","id":"launch-anthropic-opus-5-5-card-2286","ingestRunId":"launch-anthropic-opus-5-5-card","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-22","rawPayloadHash":"7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378","report":{"category":"agentic","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Strict pass rate, pass@1; 108 tasks; five runs; 1080p; 500 action steps; max reasoning effort; Opus 4.8 grader where required. September 10, 2026 task files and server-side context compaction after 100k tokens.","family":"OSWorld","locator":"Table 8.1.A; section 8.13.3","metric":"percent","notes":"Provider-published lab self-report from the Claude Opus 5.5 system card (SHA256 7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378). Not an independent board. Table 8.1.A prints partial/strict as 81.8/48.7. This row is the strict pass rate.","sourceId":"anthropic-opus-5-5-card","sourceTitle":"Claude Opus 5.5 System Card"},"score":48.7,"scoreUnit":"percent","sourceUrl":"https://www-cdn.anthropic.com/fc1b44717c85dc068bc6ba5024219938094694bd/Claude%20Opus%205.5%20System%20Card.pdf"},{"benchmarkId":"officeqa","evidenceKind":"lab_self_report","harnessId":"officeqa:anthropic-opus-5-5-card:QWdlbnRpYyBleHRyYWN0ZWQtdGV4dCBUcmVhc3VyeSBCdWxsZXRpbiBjb3JwdXMgd2l0aCBjb2RlIGV4ZWN1dGlvbjsgbWF4IGVmZm9ydDsgbWVhbiBvZiBmaXZlIHJ1bnMu","id":"launch-anthropic-opus-5-5-card-2287","ingestRunId":"launch-anthropic-opus-5-5-card","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-22","rawPayloadHash":"7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378","report":{"category":"knowledge","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Agentic extracted-text Treasury Bulletin corpus with code execution; max effort; mean of five runs.","family":"OfficeQA","locator":"section 8.14.1","metric":"percent","notes":"Provider-published lab self-report from the Claude Opus 5.5 system card (SHA256 7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378). Not an independent board.","sourceId":"anthropic-opus-5-5-card","sourceTitle":"Claude Opus 5.5 System Card"},"score":78.9,"scoreUnit":"percent","sourceUrl":"https://www-cdn.anthropic.com/fc1b44717c85dc068bc6ba5024219938094694bd/Claude%20Opus%205.5%20System%20Card.pdf"},{"benchmarkId":"officeqa-pro","evidenceKind":"lab_self_report","harnessId":"officeqa-pro:anthropic-opus-5-5-card:SGFyZGVyIDEzMy1xdWVzdGlvbiBzdWJzZXQ7IGFnZW50aWMgZXh0cmFjdGVkLXRleHQgVHJlYXN1cnkgQnVsbGV0aW4gY29ycHVzIHdpdGggY29kZSBleGVjdXRpb247IG1heCBlZmZvcnQ7IG1lYW4gb2YgZml2ZSBydW5zLg","id":"launch-anthropic-opus-5-5-card-2288","ingestRunId":"launch-anthropic-opus-5-5-card","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-22","rawPayloadHash":"7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378","report":{"category":"knowledge","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Harder 133-question subset; agentic extracted-text Treasury Bulletin corpus with code execution; max effort; mean of five runs.","family":"OfficeQA Pro","locator":"section 8.14.1","metric":"percent","notes":"Provider-published lab self-report from the Claude Opus 5.5 system card (SHA256 7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378). Not an independent board.","sourceId":"anthropic-opus-5-5-card","sourceTitle":"Claude Opus 5.5 System Card"},"score":67.7,"scoreUnit":"percent","sourceUrl":"https://www-cdn.anthropic.com/fc1b44717c85dc068bc6ba5024219938094694bd/Claude%20Opus%205.5%20System%20Card.pdf"},{"benchmarkId":"legal-agent-benchmark-120-task-held-out-subset","evidenceKind":"lab_self_report","harnessId":"legal-agent-benchmark-120-task-held-out-subset:anthropic-opus-5-5-card:QWxsLXBhc3MgcmF0ZSBvbiBIYXJ2ZXkncyBoZWxkLW91dCAxMjAgdGFza3M7IG1heCBlZmZvcnQuIEFydGlmaWNpYWwgQW5hbHlzaXMgaGFybmVzcywgYXMgaW4gdGhlIHByZXZpb3VzIHN5c3RlbSBjYXJkLg","id":"launch-anthropic-opus-5-5-card-2289","ingestRunId":"launch-anthropic-opus-5-5-card","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-22","rawPayloadHash":"7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378","report":{"category":"agentic","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"All-pass rate on Harvey's held-out 120 tasks; max effort. Artificial Analysis harness, as in the previous system card.","family":"Legal Agent Benchmark","locator":"section 8.14.2","metric":"percent","notes":"Provider-published lab self-report from the Claude Opus 5.5 system card (SHA256 7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378). Not an independent board.","sourceId":"anthropic-opus-5-5-card","sourceTitle":"Claude Opus 5.5 System Card"},"score":8.3,"scoreUnit":"percent","sourceUrl":"https://www-cdn.anthropic.com/fc1b44717c85dc068bc6ba5024219938094694bd/Claude%20Opus%205.5%20System%20Card.pdf"},{"benchmarkId":"legal-agent-benchmark-120-task-held-out-subset","evidenceKind":"lab_self_report","harnessId":"legal-agent-benchmark-120-task-held-out-subset:anthropic-opus-5-5-card:TWVhbiBjcml0ZXJpb24tcGFzcyByYXRlIG9uIEhhcnZleSdzIGhlbGQtb3V0IDEyMCB0YXNrczsgbWF4IGVmZm9ydC4gQXJ0aWZpY2lhbCBBbmFseXNpcyBoYXJuZXNzLCBhcyBpbiB0aGUgcHJldmlvdXMgc3lzdGVtIGNhcmQu","id":"launch-anthropic-opus-5-5-card-2290","ingestRunId":"launch-anthropic-opus-5-5-card","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-22","rawPayloadHash":"7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378","report":{"category":"agentic","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Mean criterion-pass rate on Harvey's held-out 120 tasks; max effort. Artificial Analysis harness, as in the previous system card.","family":"Legal Agent Benchmark","locator":"section 8.14.2","metric":"percent","notes":"Provider-published lab self-report from the Claude Opus 5.5 system card (SHA256 7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378). Not an independent board.","sourceId":"anthropic-opus-5-5-card","sourceTitle":"Claude Opus 5.5 System Card"},"score":91.2,"scoreUnit":"percent","sourceUrl":"https://www-cdn.anthropic.com/fc1b44717c85dc068bc6ba5024219938094694bd/Claude%20Opus%205.5%20System%20Card.pdf"},{"benchmarkId":"gdpval-aa-2-1","evidenceKind":"lab_self_report","harnessId":"gdpval-aa-2-1:anthropic-opus-5-5-card:QXJ0aWZpY2lhbCBBbmFseXNpcyBHRFB2YWwtQUEgdjIuMTsgMjIwIEdEUHZhbCBnb2xkIHRhc2tzOyBibGluZCBwYWlyd2lzZSBFbG8gYW5jaG9yZWQgdG8gRGVlcFNlZWsgVjQuMSBGbGFzaCAobWF4KSBhdCAxNjAwOyBtYXggZWZmb3J0LiBSdW4gYnkgQXJ0aWZpY2lhbCBBbmFseXNpcy4","id":"launch-anthropic-opus-5-5-card-2291","ingestRunId":"launch-anthropic-opus-5-5-card","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-22","rawPayloadHash":"7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378","report":{"category":"human_pref","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Artificial Analysis GDPval-AA v2.1; 220 GDPval gold tasks; blind pairwise Elo anchored to DeepSeek V4.1 Flash (max) at 1600; max effort. Run by Artificial Analysis.","family":"GDPval-AA","locator":"Table 8.1.A; section 8.14.3","metric":"Elo","notes":"Provider-published lab self-report from the Claude Opus 5.5 system card (SHA256 7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378). Not an independent board. Launch grid shows 1846.","sourceId":"anthropic-opus-5-5-card","sourceTitle":"Claude Opus 5.5 System Card"},"score":1846,"scoreUnit":"elo","sourceUrl":"https://www-cdn.anthropic.com/fc1b44717c85dc068bc6ba5024219938094694bd/Claude%20Opus%205.5%20System%20Card.pdf"},{"benchmarkId":"gdpval-aa-2-1","evidenceKind":"lab_self_report","harnessId":"gdpval-aa-2-1:anthropic-opus-5-5-card:QXJ0aWZpY2lhbCBBbmFseXNpcyBHRFB2YWwtQUEgdjIuMTsgMjIwIEdEUHZhbCBnb2xkIHRhc2tzOyBibGluZCBwYWlyd2lzZSBFbG8gYW5jaG9yZWQgdG8gRGVlcFNlZWsgVjQuMSBGbGFzaCAobWF4KSBhdCAxNjAwOyB4aGlnaCBlZmZvcnQuIFJ1biBieSBBcnRpZmljaWFsIEFuYWx5c2lzLg","id":"launch-anthropic-opus-5-5-card-2292","ingestRunId":"launch-anthropic-opus-5-5-card","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-22","rawPayloadHash":"7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378","report":{"category":"human_pref","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Artificial Analysis GDPval-AA v2.1; 220 GDPval gold tasks; blind pairwise Elo anchored to DeepSeek V4.1 Flash (max) at 1600; xhigh effort. Run by Artificial Analysis.","family":"GDPval-AA","locator":"section 8.14.3","metric":"Elo","notes":"Provider-published lab self-report from the Claude Opus 5.5 system card (SHA256 7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378). Not an independent board.","sourceId":"anthropic-opus-5-5-card","sourceTitle":"Claude Opus 5.5 System Card"},"score":1820,"scoreUnit":"elo","sourceUrl":"https://www-cdn.anthropic.com/fc1b44717c85dc068bc6ba5024219938094694bd/Claude%20Opus%205.5%20System%20Card.pdf"},{"benchmarkId":"aa-briefcase-1-1-elo","evidenceKind":"lab_self_report","harnessId":"aa-briefcase-1-1-elo:anthropic-opus-5-5-card:QXJ0aWZpY2lhbCBBbmFseXNpcyBBQS1CcmllZmNhc2UgdjEuMTsgbG9uZy1ob3Jpem9uIGtub3dsZWRnZSBwcm9qZWN0czsgcnVicmljIHNjb3JpbmcgYW5kIHBhaXJ3aXNlIGp1ZGdpbmc7IG1heCBlZmZvcnQuIFJ1biBieSBBcnRpZmljaWFsIEFuYWx5c2lzLg","id":"launch-anthropic-opus-5-5-card-2293","ingestRunId":"launch-anthropic-opus-5-5-card","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-22","rawPayloadHash":"7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378","report":{"category":"human_pref","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Artificial Analysis AA-Briefcase v1.1; long-horizon knowledge projects; rubric scoring and pairwise judging; max effort. Run by Artificial Analysis.","family":"AA-Briefcase","locator":"Table 8.1.A; section 8.14.4","metric":"Elo","notes":"Provider-published lab self-report from the Claude Opus 5.5 system card (SHA256 7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378). Not an independent board.","sourceId":"anthropic-opus-5-5-card","sourceTitle":"Claude Opus 5.5 System Card"},"score":1822,"scoreUnit":"elo","sourceUrl":"https://www-cdn.anthropic.com/fc1b44717c85dc068bc6ba5024219938094694bd/Claude%20Opus%205.5%20System%20Card.pdf"},{"benchmarkId":"aa-briefcase-1-1-elo","evidenceKind":"lab_self_report","harnessId":"aa-briefcase-1-1-elo:anthropic-opus-5-5-card:QXJ0aWZpY2lhbCBBbmFseXNpcyBBQS1CcmllZmNhc2UgdjEuMTsgbG9uZy1ob3Jpem9uIGtub3dsZWRnZSBwcm9qZWN0czsgcnVicmljIHNjb3JpbmcgYW5kIHBhaXJ3aXNlIGp1ZGdpbmc7IHhoaWdoIGVmZm9ydC4gUnVuIGJ5IEFydGlmaWNpYWwgQW5hbHlzaXMu","id":"launch-anthropic-opus-5-5-card-2294","ingestRunId":"launch-anthropic-opus-5-5-card","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-22","rawPayloadHash":"7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378","report":{"category":"human_pref","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Artificial Analysis AA-Briefcase v1.1; long-horizon knowledge projects; rubric scoring and pairwise judging; xhigh effort. Run by Artificial Analysis.","family":"AA-Briefcase","locator":"section 8.14.4","metric":"Elo","notes":"Provider-published lab self-report from the Claude Opus 5.5 system card (SHA256 7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378). Not an independent board.","sourceId":"anthropic-opus-5-5-card","sourceTitle":"Claude Opus 5.5 System Card"},"score":1780,"scoreUnit":"elo","sourceUrl":"https://www-cdn.anthropic.com/fc1b44717c85dc068bc6ba5024219938094694bd/Claude%20Opus%205.5%20System%20Card.pdf"},{"benchmarkId":"aa-briefcase-1-1-elo","evidenceKind":"lab_self_report","harnessId":"aa-briefcase-1-1-elo:anthropic-opus-5-5-card:QXJ0aWZpY2lhbCBBbmFseXNpcyBBQS1CcmllZmNhc2UgdjEuMTsgbG9uZy1ob3Jpem9uIGtub3dsZWRnZSBwcm9qZWN0czsgcnVicmljIHNjb3JpbmcgYW5kIHBhaXJ3aXNlIGp1ZGdpbmc7IGhpZ2ggZWZmb3J0LiBSdW4gYnkgQXJ0aWZpY2lhbCBBbmFseXNpcy4","id":"launch-anthropic-opus-5-5-card-2295","ingestRunId":"launch-anthropic-opus-5-5-card","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-22","rawPayloadHash":"7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378","report":{"category":"human_pref","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Artificial Analysis AA-Briefcase v1.1; long-horizon knowledge projects; rubric scoring and pairwise judging; high effort. Run by Artificial Analysis.","family":"AA-Briefcase","locator":"section 8.14.4","metric":"Elo","notes":"Provider-published lab self-report from the Claude Opus 5.5 system card (SHA256 7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378). Not an independent board.","sourceId":"anthropic-opus-5-5-card","sourceTitle":"Claude Opus 5.5 System Card"},"score":1705,"scoreUnit":"elo","sourceUrl":"https://www-cdn.anthropic.com/fc1b44717c85dc068bc6ba5024219938094694bd/Claude%20Opus%205.5%20System%20Card.pdf"},{"benchmarkId":"toolathlon-verified-june2026","evidenceKind":"lab_self_report","harnessId":"toolathlon-verified-june2026:anthropic-opus-5-5-card:UGFzc0AxOyAxMDggdGFza3M7IHRocmVlIHRyaWFsczsgaW50ZXJuYWwgaGFybmVzcyBtaXJyb3JpbmcgVG9vbGF0aGxvbi1WZXJpZmllZDsgYWRhcHRpdmUgdGhpbmtpbmcgYXQgbWF4IGVmZm9ydDsgc2FmZXR5IGNsYXNzaWZpZXJzIG9uOyBvbmUgc2FmZXR5IHN0b3AgYW5kIHNpeCBzYW5kYm94LW1vbml0b3IgaGFsdHMgY291bnRlZCBhcyBmYWlsdXJlcy4","id":"launch-anthropic-opus-5-5-card-2296","ingestRunId":"launch-anthropic-opus-5-5-card","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-22","rawPayloadHash":"7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378","report":{"category":"agentic","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Pass@1; 108 tasks; three trials; internal harness mirroring Toolathlon-Verified; adaptive thinking at max effort; safety classifiers on; one safety stop and six sandbox-monitor halts counted as failures.","family":"Toolathlon","locator":"Table 8.14.5.A","metric":"percent","notes":"Provider-published lab self-report from the Claude Opus 5.5 system card (SHA256 7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378). Not an independent board.","sourceId":"anthropic-opus-5-5-card","sourceTitle":"Claude Opus 5.5 System Card"},"score":77.8,"scoreUnit":"percent","sourceUrl":"https://www-cdn.anthropic.com/fc1b44717c85dc068bc6ba5024219938094694bd/Claude%20Opus%205.5%20System%20Card.pdf"},{"benchmarkId":"toolathlon-verified-june2026","evidenceKind":"lab_self_report","harnessId":"toolathlon-verified-june2026:anthropic-opus-5-5-card:UGFzc0AzIChhdCBsZWFzdCBvbmUgb2YgdGhyZWUgdHJpYWxzIGNvcnJlY3QpOyAxMDggdGFza3M7IGludGVybmFsIGhhcm5lc3M7IGFkYXB0aXZlIHRoaW5raW5nIGF0IG1heCBlZmZvcnQ7IHNhZmV0eSBjbGFzc2lmaWVycyBvbi4","id":"launch-anthropic-opus-5-5-card-2297","ingestRunId":"launch-anthropic-opus-5-5-card","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-22","rawPayloadHash":"7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378","report":{"category":"agentic","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Pass@3 (at least one of three trials correct); 108 tasks; internal harness; adaptive thinking at max effort; safety classifiers on.","family":"Toolathlon","locator":"Table 8.14.5.A","metric":"percent","notes":"Provider-published lab self-report from the Claude Opus 5.5 system card (SHA256 7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378). Not an independent board.","sourceId":"anthropic-opus-5-5-card","sourceTitle":"Claude Opus 5.5 System Card"},"score":82.4,"scoreUnit":"percent","sourceUrl":"https://www-cdn.anthropic.com/fc1b44717c85dc068bc6ba5024219938094694bd/Claude%20Opus%205.5%20System%20Card.pdf"},{"benchmarkId":"toolathlon-verified-june2026","evidenceKind":"lab_self_report","harnessId":"toolathlon-verified-june2026:anthropic-opus-5-5-card:UGFzc8KzIChhbGwgdGhyZWUgdHJpYWxzIGNvcnJlY3QpOyAxMDggdGFza3M7IGludGVybmFsIGhhcm5lc3M7IGFkYXB0aXZlIHRoaW5raW5nIGF0IG1heCBlZmZvcnQ7IHNhZmV0eSBjbGFzc2lmaWVycyBvbi4","id":"launch-anthropic-opus-5-5-card-2298","ingestRunId":"launch-anthropic-opus-5-5-card","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-22","rawPayloadHash":"7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378","report":{"category":"agentic","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Pass³ (all three trials correct); 108 tasks; internal harness; adaptive thinking at max effort; safety classifiers on.","family":"Toolathlon","locator":"Table 8.14.5.A","metric":"percent","notes":"Provider-published lab self-report from the Claude Opus 5.5 system card (SHA256 7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378). Not an independent board.","sourceId":"anthropic-opus-5-5-card","sourceTitle":"Claude Opus 5.5 System Card"},"score":72.2,"scoreUnit":"percent","sourceUrl":"https://www-cdn.anthropic.com/fc1b44717c85dc068bc6ba5024219938094694bd/Claude%20Opus%205.5%20System%20Card.pdf"},{"benchmarkId":"automationbench","evidenceKind":"lab_self_report","harnessId":"automationbench:anthropic-opus-5-5-card:WmFwaWVyIHByaXZhdGUgaGVsZC1vdXQgbGVhZGVyYm9hcmQ7IHNpbXVsYXRlZCBidXNpbmVzcyB3b3JrZmxvd3M7IGV2ZXJ5IGRldGVybWluaXN0aWMgYXNzZXJ0aW9uIG11c3QgcGFzczsgbWF4IGVmZm9ydC4","id":"launch-anthropic-opus-5-5-card-2299","ingestRunId":"launch-anthropic-opus-5-5-card","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-22","rawPayloadHash":"7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378","report":{"category":"agentic","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Zapier private held-out leaderboard; simulated business workflows; every deterministic assertion must pass; max effort.","family":"AutomationBench","locator":"Table 8.1.A; section 8.14.6","metric":"percent","notes":"Provider-published lab self-report from the Claude Opus 5.5 system card (SHA256 7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378). Not an independent board. Launch grid shows 40.0%. Launch footnote 2 says Zapier ran these without fallback models and counted safeguard interventions as failures.","sourceId":"anthropic-opus-5-5-card","sourceTitle":"Claude Opus 5.5 System Card"},"score":40,"scoreUnit":"percent","sourceUrl":"https://www-cdn.anthropic.com/fc1b44717c85dc068bc6ba5024219938094694bd/Claude%20Opus%205.5%20System%20Card.pdf"},{"benchmarkId":"healthbench","evidenceKind":"lab_self_report","harnessId":"healthbench:anthropic-opus-5-5-card:UmF3IHJ1YnJpYyBzY29yZTsgYWRhcHRpdmUgdGhpbmtpbmcgYXQgbWF4IGVmZm9ydDsgZml2ZSB0cmlhbHM7IG5vIHRvb2xzIG9yIGN1c3RvbSBzeXN0ZW0gcHJvbXB0OyBPcHVzIDQuOCBncmFkZXI7IHNhZmV0eSBjbGFzc2lmaWVycyB3aXRoIHJlZnVzYWwgZmFsbGJhY2sgdG8gT3B1cyA1Lg","id":"launch-anthropic-opus-5-5-card-2300","ingestRunId":"launch-anthropic-opus-5-5-card","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-22","rawPayloadHash":"7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378","report":{"category":"knowledge","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Raw rubric score; adaptive thinking at max effort; five trials; no tools or custom system prompt; Opus 4.8 grader; safety classifiers with refusal fallback to Opus 5.","family":"HealthBench","locator":"section 8.15.1","metric":"percent","notes":"Provider-published lab self-report from the Claude Opus 5.5 system card (SHA256 7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378). Not an independent board.","sourceId":"anthropic-opus-5-5-card","sourceTitle":"Claude Opus 5.5 System Card"},"score":68.1,"scoreUnit":"percent","sourceUrl":"https://www-cdn.anthropic.com/fc1b44717c85dc068bc6ba5024219938094694bd/Claude%20Opus%205.5%20System%20Card.pdf"},{"benchmarkId":"healthbench","evidenceKind":"lab_self_report","harnessId":"healthbench:anthropic-opus-5-5-card:TGVuZ3RoLWFkanVzdGVkIHNjb3JlIHVzaW5nIHRoZSBHUFQtNS41IHN5c3RlbS1jYXJkIG1ldGhvZDsgb3RoZXJ3aXNlIHRoZSByYXcgSGVhbHRoQmVuY2ggY29uZmlndXJhdGlvbjogYWRhcHRpdmUgbWF4LCBmaXZlIHRyaWFscywgbm8gdG9vbHMsIE9wdXMgNC44IGdyYWRlciwgc2FmZXR5IGZhbGxiYWNrIHRvIE9wdXMgNS4","id":"launch-anthropic-opus-5-5-card-2301","ingestRunId":"launch-anthropic-opus-5-5-card","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-22","rawPayloadHash":"7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378","report":{"category":"knowledge","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Length-adjusted score using the GPT-5.5 system-card method; otherwise the raw HealthBench configuration: adaptive max, five trials, no tools, Opus 4.8 grader, safety fallback to Opus 5.","family":"HealthBench","locator":"section 8.15.1","metric":"percent","notes":"Provider-published lab self-report from the Claude Opus 5.5 system card (SHA256 7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378). Not an independent board.","sourceId":"anthropic-opus-5-5-card","sourceTitle":"Claude Opus 5.5 System Card"},"score":60.6,"scoreUnit":"percent","sourceUrl":"https://www-cdn.anthropic.com/fc1b44717c85dc068bc6ba5024219938094694bd/Claude%20Opus%205.5%20System%20Card.pdf"},{"benchmarkId":"healthbench-professional-percent","evidenceKind":"lab_self_report","harnessId":"healthbench-professional-percent:anthropic-opus-5-5-card:UmF3IHJ1YnJpYyBzY29yZTsgYWRhcHRpdmUgdGhpbmtpbmcgYXQgbWF4IGVmZm9ydDsgZml2ZSB0cmlhbHM7IG5vIHRvb2xzIG9yIGN1c3RvbSBzeXN0ZW0gcHJvbXB0OyBPcHVzIDQuOCBncmFkZXI7IHNhZmV0eSBjbGFzc2lmaWVycyB3aXRoIHJlZnVzYWwgZmFsbGJhY2sgdG8gT3B1cyA1Lg","id":"launch-anthropic-opus-5-5-card-2302","ingestRunId":"launch-anthropic-opus-5-5-card","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-22","rawPayloadHash":"7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378","report":{"category":"knowledge","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Raw rubric score; adaptive thinking at max effort; five trials; no tools or custom system prompt; Opus 4.8 grader; safety classifiers with refusal fallback to Opus 5.","family":"HealthBench Professional","locator":"section 8.15.2","metric":"percent","notes":"Provider-published lab self-report from the Claude Opus 5.5 system card (SHA256 7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378). Not an independent board.","sourceId":"anthropic-opus-5-5-card","sourceTitle":"Claude Opus 5.5 System Card"},"score":77.1,"scoreUnit":"percent","sourceUrl":"https://www-cdn.anthropic.com/fc1b44717c85dc068bc6ba5024219938094694bd/Claude%20Opus%205.5%20System%20Card.pdf"},{"benchmarkId":"healthbench-professional-percent","evidenceKind":"lab_self_report","harnessId":"healthbench-professional-percent:anthropic-opus-5-5-card:TGVuZ3RoLWFkanVzdGVkIHNjb3JlIHVzaW5nIHRoZSBIZWFsdGhCZW5jaCBQcm9mZXNzaW9uYWwgcGFwZXIgbWV0aG9kOyBhZGFwdGl2ZSBtYXg7IGZpdmUgdHJpYWxzOyBubyB0b29sczsgT3B1cyA0LjggZ3JhZGVyOyBzYWZldHkgZmFsbGJhY2sgdG8gT3B1cyA1Lg","id":"launch-anthropic-opus-5-5-card-2303","ingestRunId":"launch-anthropic-opus-5-5-card","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-22","rawPayloadHash":"7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378","report":{"category":"knowledge","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Length-adjusted score using the HealthBench Professional paper method; adaptive max; five trials; no tools; Opus 4.8 grader; safety fallback to Opus 5.","family":"HealthBench Professional","locator":"Table 8.1.A; section 8.15.2","metric":"percent","notes":"Provider-published lab self-report from the Claude Opus 5.5 system card (SHA256 7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378). Not an independent board. Table 8.1.A's 65.6 is this length-adjusted score, not the raw 77.1%.","sourceId":"anthropic-opus-5-5-card","sourceTitle":"Claude Opus 5.5 System Card"},"score":65.6,"scoreUnit":"percent","sourceUrl":"https://www-cdn.anthropic.com/fc1b44717c85dc068bc6ba5024219938094694bd/Claude%20Opus%205.5%20System%20Card.pdf"},{"benchmarkId":"gmmlu","evidenceKind":"lab_self_report","harnessId":"gmmlu:anthropic-opus-5-5-card:QXZlcmFnZSBhY2N1cmFjeSBhY3Jvc3MgNDIgbGFuZ3VhZ2VzOyBhZGFwdGl2ZSB0aGlua2luZyBhdCBtYXggZWZmb3J0OyBzaW5nbGUgdHJpYWw7IG5vIHRvb2xzIG9yIGN1c3RvbSBzeXN0ZW0gcHJvbXB0Lg","id":"launch-anthropic-opus-5-5-card-2304","ingestRunId":"launch-anthropic-opus-5-5-card","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-22","rawPayloadHash":"7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378","report":{"category":"knowledge","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Average accuracy across 42 languages; adaptive thinking at max effort; single trial; no tools or custom system prompt.","family":"GMMLU","locator":"section 8.16.1; Figure 8.16.1.A","metric":"percent","notes":"Provider-published lab self-report from the Claude Opus 5.5 system card (SHA256 7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378). Not an independent board.","sourceId":"anthropic-opus-5-5-card","sourceTitle":"Claude Opus 5.5 System Card"},"score":94.3,"scoreUnit":"percent","sourceUrl":"https://www-cdn.anthropic.com/fc1b44717c85dc068bc6ba5024219938094694bd/Claude%20Opus%205.5%20System%20Card.pdf"},{"benchmarkId":"milu","evidenceKind":"lab_self_report","harnessId":"milu:anthropic-opus-5-5-card:QXZlcmFnZSBhY2N1cmFjeSBhY3Jvc3MgMTEgbGFuZ3VhZ2VzOyBhZGFwdGl2ZSB0aGlua2luZyBhdCBtYXggZWZmb3J0OyBmaXZlIHRyaWFsczsgbm8gdG9vbHMgb3IgY3VzdG9tIHN5c3RlbSBwcm9tcHQu","id":"launch-anthropic-opus-5-5-card-2305","ingestRunId":"launch-anthropic-opus-5-5-card","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-22","rawPayloadHash":"7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378","report":{"category":"knowledge","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Average accuracy across 11 languages; adaptive thinking at max effort; five trials; no tools or custom system prompt.","family":"MILU","locator":"section 8.16.2; Figure 8.16.2.A","metric":"percent","notes":"Provider-published lab self-report from the Claude Opus 5.5 system card (SHA256 7311c9c6bbb16d012f1c12c7418b05949fcf7ae3e30d2c40f22050074b2a7378). Not an independent board.","sourceId":"anthropic-opus-5-5-card","sourceTitle":"Claude Opus 5.5 System Card"},"score":93.1,"scoreUnit":"percent","sourceUrl":"https://www-cdn.anthropic.com/fc1b44717c85dc068bc6ba5024219938094694bd/Claude%20Opus%205.5%20System%20Card.pdf"},{"benchmarkId":"tau2-bench-telecom-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"tau2-bench-telecom-source-release-snapshot-version-not-specified:cohere:Q29oZXJlIGxhdW5jaCBldmFsdWF0aW9uOyBOb3J0aCBtZXRyaWMgdXNlcyBpbnRlcm5hbCBMTE0ganVkZ2Uu","id":"launch-cohere-1429","ingestRunId":"launch-cohere","modelId":"command-a-plus-05-2026","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Cohere launch evaluation; North metric uses internal LLM judge.","family":"tau2-Bench Telecom","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"cohere","sourceTitle":"Command A+ launch benchmarks"},"score":85,"scoreUnit":"percent","sourceUrl":"https://cohere.com/blog/command-a-plus"},{"benchmarkId":"tau2-bench-telecom-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"tau2-bench-telecom-source-release-snapshot-version-not-specified:cohere:Q29oZXJlIGxhdW5jaCBldmFsdWF0aW9uOyBOb3J0aCBtZXRyaWMgdXNlcyBpbnRlcm5hbCBMTE0ganVkZ2Uu","id":"launch-cohere-1430","ingestRunId":"launch-cohere","modelId":"command-a-reasoning-08-2025","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Cohere launch evaluation; North metric uses internal LLM judge.","family":"tau2-Bench Telecom","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"cohere","sourceTitle":"Command A+ launch benchmarks"},"score":37,"scoreUnit":"percent","sourceUrl":"https://cohere.com/blog/command-a-plus"},{"benchmarkId":"terminal-bench-hard-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"terminal-bench-hard-source-release-snapshot-version-not-specified:cohere:Q29oZXJlIGxhdW5jaCBldmFsdWF0aW9uOyBOb3J0aCBtZXRyaWMgdXNlcyBpbnRlcm5hbCBMTE0ganVkZ2Uu","id":"launch-cohere-1431","ingestRunId":"launch-cohere","modelId":"command-a-plus-05-2026","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Cohere launch evaluation; North metric uses internal LLM judge.","family":"Terminal-Bench","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"cohere","sourceTitle":"Command A+ launch benchmarks"},"score":25,"scoreUnit":"percent","sourceUrl":"https://cohere.com/blog/command-a-plus"},{"benchmarkId":"terminal-bench-hard-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"terminal-bench-hard-source-release-snapshot-version-not-specified:cohere:Q29oZXJlIGxhdW5jaCBldmFsdWF0aW9uOyBOb3J0aCBtZXRyaWMgdXNlcyBpbnRlcm5hbCBMTE0ganVkZ2Uu","id":"launch-cohere-1432","ingestRunId":"launch-cohere","modelId":"command-a-reasoning-08-2025","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Cohere launch evaluation; North metric uses internal LLM judge.","family":"Terminal-Bench","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"cohere","sourceTitle":"Command A+ launch benchmarks"},"score":3,"scoreUnit":"percent","sourceUrl":"https://cohere.com/blog/command-a-plus"},{"benchmarkId":"north-memory-usage-quality-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"north-memory-usage-quality-source-release-snapshot-version-not-specified:cohere:Q29oZXJlIGxhdW5jaCBldmFsdWF0aW9uOyBOb3J0aCBtZXRyaWMgdXNlcyBpbnRlcm5hbCBMTE0ganVkZ2Uu","id":"launch-cohere-1433","ingestRunId":"launch-cohere","modelId":"command-a-plus-05-2026","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Cohere launch evaluation; North metric uses internal LLM judge.","family":"North Memory Usage Quality","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"cohere","sourceTitle":"Command A+ launch benchmarks"},"score":54,"scoreUnit":"percent","sourceUrl":"https://cohere.com/blog/command-a-plus"},{"benchmarkId":"north-memory-usage-quality-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"north-memory-usage-quality-source-release-snapshot-version-not-specified:cohere:Q29oZXJlIGxhdW5jaCBldmFsdWF0aW9uOyBOb3J0aCBtZXRyaWMgdXNlcyBpbnRlcm5hbCBMTE0ganVkZ2Uu","id":"launch-cohere-1434","ingestRunId":"launch-cohere","modelId":"command-a-reasoning-08-2025","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Cohere launch evaluation; North metric uses internal LLM judge.","family":"North Memory Usage Quality","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"cohere","sourceTitle":"Command A+ launch benchmarks"},"score":39,"scoreUnit":"percent","sourceUrl":"https://cohere.com/blog/command-a-plus"},{"benchmarkId":"mmmu-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mmmu-pro-source-release-snapshot-version-not-specified:cohere:Q29oZXJlIGxhdW5jaCBtdWx0aW1vZGFsIGV2YWx1YXRpb25zLg","id":"launch-cohere-1435","ingestRunId":"launch-cohere","modelId":"command-a-plus-05-2026","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Cohere launch multimodal evaluations.","family":"MMMU-Pro","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"cohere","sourceTitle":"Command A+ launch benchmarks"},"score":63,"scoreUnit":"percent","sourceUrl":"https://cohere.com/blog/command-a-plus"},{"benchmarkId":"mmmu-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mmmu-source-release-snapshot-version-not-specified:cohere:Q29oZXJlIGxhdW5jaCBtdWx0aW1vZGFsIGV2YWx1YXRpb25zLg","id":"launch-cohere-1436","ingestRunId":"launch-cohere","modelId":"command-a-plus-05-2026","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Cohere launch multimodal evaluations.","family":"MMMU","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"cohere","sourceTitle":"Command A+ launch benchmarks"},"score":75.1,"scoreUnit":"percent","sourceUrl":"https://cohere.com/blog/command-a-plus"},{"benchmarkId":"mmmu-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mmmu-source-release-snapshot-version-not-specified:cohere:Q29oZXJlIGxhdW5jaCBtdWx0aW1vZGFsIGV2YWx1YXRpb25zLg","id":"launch-cohere-1437","ingestRunId":"launch-cohere","modelId":"command-a-vision-07-2025","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Cohere launch multimodal evaluations.","family":"MMMU","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"cohere","sourceTitle":"Command A+ launch benchmarks"},"score":65.3,"scoreUnit":"percent","sourceUrl":"https://cohere.com/blog/command-a-plus"},{"benchmarkId":"mathvista-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mathvista-source-release-snapshot-version-not-specified:cohere:Q29oZXJlIGxhdW5jaCBtdWx0aW1vZGFsIGV2YWx1YXRpb25zLg","id":"launch-cohere-1438","ingestRunId":"launch-cohere","modelId":"command-a-plus-05-2026","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Cohere launch multimodal evaluations.","family":"MathVista","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"cohere","sourceTitle":"Command A+ launch benchmarks"},"score":80.6,"scoreUnit":"percent","sourceUrl":"https://cohere.com/blog/command-a-plus"},{"benchmarkId":"mathvista-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mathvista-source-release-snapshot-version-not-specified:cohere:Q29oZXJlIGxhdW5jaCBtdWx0aW1vZGFsIGV2YWx1YXRpb25zLg","id":"launch-cohere-1439","ingestRunId":"launch-cohere","modelId":"command-a-vision-07-2025","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Cohere launch multimodal evaluations.","family":"MathVista","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"cohere","sourceTitle":"Command A+ launch benchmarks"},"score":73.5,"scoreUnit":"percent","sourceUrl":"https://cohere.com/blog/command-a-plus"},{"benchmarkId":"charxiv-reasoning-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"charxiv-reasoning-source-release-snapshot-version-not-specified:cohere:Q29oZXJlIGxhdW5jaCBtdWx0aW1vZGFsIGV2YWx1YXRpb25zLg","id":"launch-cohere-1440","ingestRunId":"launch-cohere","modelId":"command-a-plus-05-2026","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Cohere launch multimodal evaluations.","family":"CharXiv Reasoning","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"cohere","sourceTitle":"Command A+ launch benchmarks"},"score":52.7,"scoreUnit":"percent","sourceUrl":"https://cohere.com/blog/command-a-plus"},{"benchmarkId":"charxiv-reasoning-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"charxiv-reasoning-source-release-snapshot-version-not-specified:cohere:Q29oZXJlIGxhdW5jaCBtdWx0aW1vZGFsIGV2YWx1YXRpb25zLg","id":"launch-cohere-1441","ingestRunId":"launch-cohere","modelId":"command-a-vision-07-2025","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Cohere launch multimodal evaluations.","family":"CharXiv Reasoning","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"cohere","sourceTitle":"Command A+ launch benchmarks"},"score":46.9,"scoreUnit":"percent","sourceUrl":"https://cohere.com/blog/command-a-plus"},{"benchmarkId":"ifbench-source-release-snapshot-version-as-labeled","evidenceKind":"lab_self_report","harnessId":"ifbench-source-release-snapshot-version-as-labeled:cohere:U2luZ2xlLXR1cm4gbG9vc2UscHJvbXB0IGFjY3VyYWN5LDI5NHByb21wdHMgeDUgcmVwZWF0cy4","id":"launch-cohere-1967","ingestRunId":"launch-cohere","modelId":"command-a-plus-05-2026","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"Single-turn loose,prompt accuracy,294prompts x5 repeats.","family":"IFBench","locator":"Image3","metric":"%","notes":"First-party chart numeric label, visually verified.","sourceId":"cohere","sourceTitle":"Command A+ launch benchmarks"},"score":74,"scoreUnit":"percent","sourceUrl":"https://cohere.com/blog/command-a-plus"},{"benchmarkId":"aime-2025-source-release-snapshot-version-as-labeled","evidenceKind":"lab_self_report","harnessId":"aime-2025-source-release-snapshot-version-as-labeled:cohere:T2ZmaWNpYWwzMHF1ZXN0aW9ucyB4MTAgcmVwZWF0czsgcGFzc0AxLg","id":"launch-cohere-1968","ingestRunId":"launch-cohere","modelId":"command-a-plus-05-2026","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Official30questions x10 repeats; pass@1.","family":"AIME","locator":"Image3","metric":"%","notes":"First-party chart numeric label, visually verified.","sourceId":"cohere","sourceTitle":"Command A+ launch benchmarks"},"score":90,"scoreUnit":"percent","sourceUrl":"https://cohere.com/blog/command-a-plus"},{"benchmarkId":"scicode-source-release-snapshot-version-as-labeled","evidenceKind":"lab_self_report","harnessId":"scicode-source-release-snapshot-version-as-labeled:cohere:NjVwcm9ibGVtcy8yODhzdWJwcm9ibGVtczsgc2NpZW50aXN0LWFubm90YXRlZCBiYWNrZ3JvdW5kLg","id":"launch-cohere-1969","ingestRunId":"launch-cohere","modelId":"command-a-plus-05-2026","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"65problems/288subproblems; scientist-annotated background.","family":"SciCode","locator":"Image3","metric":"%","notes":"First-party chart numeric label, visually verified.","sourceId":"cohere","sourceTitle":"Command A+ launch benchmarks"},"score":38,"scoreUnit":"percent","sourceUrl":"https://cohere.com/blog/command-a-plus"},{"benchmarkId":"north-agentic-question-answering-source-release-snapshot-version-as-labeled","evidenceKind":"lab_self_report","harnessId":"north-agentic-question-answering-source-release-snapshot-version-as-labeled:cohere:SW50ZXJuYWwgTm9ydGggZW50ZXJwcmlzZSBNQ1AgY2xvdWQtZmlsZSBRQSxMTE0ganVkZ2Uu","id":"launch-cohere-1970","ingestRunId":"launch-cohere","modelId":"command-a-plus-05-2026","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Internal North enterprise MCP cloud-file QA,LLM judge.","family":"North Agentic Question Answering","locator":"Image4","metric":"%","notes":"First-party chart numeric label, visually verified.","sourceId":"cohere","sourceTitle":"Command A+ launch benchmarks"},"score":65,"scoreUnit":"percent","sourceUrl":"https://cohere.com/blog/command-a-plus"},{"benchmarkId":"north-data-analysis-source-release-snapshot-version-as-labeled","evidenceKind":"lab_self_report","harnessId":"north-data-analysis-source-release-snapshot-version-as-labeled:cohere:SW50ZXJuYWwgTm9ydGggdXBsb2FkZWQgc3ByZWFkc2hlZXQgZGF0YS1zY2llbmNlIHRhc2tzLExMTSBqdWRnZS4","id":"launch-cohere-1971","ingestRunId":"launch-cohere","modelId":"command-a-plus-05-2026","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Internal North uploaded spreadsheet data-science tasks,LLM judge.","family":"North Data Analysis","locator":"Image4","metric":"%","notes":"First-party chart numeric label, visually verified.","sourceId":"cohere","sourceTitle":"Command A+ launch benchmarks"},"score":45,"scoreUnit":"percent","sourceUrl":"https://cohere.com/blog/command-a-plus"},{"benchmarkId":"mt-aime-2025-arabic-japanese-korean-source-release-snapshot-version-as-labeled","evidenceKind":"lab_self_report","harnessId":"mt-aime-2025-arabic-japanese-korean-source-release-snapshot-version-as-labeled:cohere:SW50ZXJuYWwgQ29tbWFuZCBBIFRyYW5zbGF0ZSB0cmFuc2xhdGlvbnM7IEFyYWJpYyxKYXBhbmVzZSxLb3JlYW4u","id":"launch-cohere-1972","ingestRunId":"launch-cohere","modelId":"command-a-plus-05-2026","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Internal Command A Translate translations; Arabic,Japanese,Korean.","family":"MT-AIME 2025 Arabic/Japanese/Korean","locator":"Image6","metric":"%","notes":"First-party chart numeric label, visually verified.","sourceId":"cohere","sourceTitle":"Command A+ launch benchmarks"},"score":86,"scoreUnit":"percent","sourceUrl":"https://cohere.com/blog/command-a-plus"},{"benchmarkId":"wmt24-50-varieties-source-release-snapshot-version-as-labeled","evidenceKind":"lab_self_report","harnessId":"wmt24-50-varieties-source-release-snapshot-version-as-labeled:cohere:eENPTUVUeGwgYXZlcmFnZTUwIHZhcmlldGllcyxpbmNsdWRpbmcgaW50ZXJuYWwgSXJpc2gvTWFsdGVzZSB0cmFuc2xhdGlvbnMgYW5kIFNlcmJpYW4gdHJhbnNsaXRlcmF0aW9uLg","id":"launch-cohere-1973","ingestRunId":"launch-cohere","modelId":"command-a-plus-05-2026","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"xCOMETxl average50 varieties,including internal Irish/Maltese translations and Serbian transliteration.","family":"WMT24++ 50 varieties","locator":"Image6","metric":"xCOMETxl score","notes":"First-party chart numeric label, visually verified.","sourceId":"cohere","sourceTitle":"Command A+ launch benchmarks"},"score":81,"scoreUnit":"index","sourceUrl":"https://cohere.com/blog/command-a-plus"},{"benchmarkId":"ifbench-source-release-snapshot-version-as-labeled","evidenceKind":"lab_self_report","harnessId":"ifbench-source-release-snapshot-version-as-labeled:cohere:U2luZ2xlLXR1cm4gbG9vc2UscHJvbXB0IGFjY3VyYWN5LDI5NHByb21wdHMgeDUgcmVwZWF0cy4","id":"launch-cohere-1974","ingestRunId":"launch-cohere","modelId":"command-a-reasoning-08-2025","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"Single-turn loose,prompt accuracy,294prompts x5 repeats.","family":"IFBench","locator":"Image3","metric":"%","notes":"First-party chart numeric label, visually verified.","sourceId":"cohere","sourceTitle":"Command A+ launch benchmarks"},"score":36,"scoreUnit":"percent","sourceUrl":"https://cohere.com/blog/command-a-plus"},{"benchmarkId":"aime-2025-source-release-snapshot-version-as-labeled","evidenceKind":"lab_self_report","harnessId":"aime-2025-source-release-snapshot-version-as-labeled:cohere:T2ZmaWNpYWwzMHF1ZXN0aW9ucyB4MTAgcmVwZWF0czsgcGFzc0AxLg","id":"launch-cohere-1975","ingestRunId":"launch-cohere","modelId":"command-a-reasoning-08-2025","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Official30questions x10 repeats; pass@1.","family":"AIME","locator":"Image3","metric":"%","notes":"First-party chart numeric label, visually verified.","sourceId":"cohere","sourceTitle":"Command A+ launch benchmarks"},"score":57,"scoreUnit":"percent","sourceUrl":"https://cohere.com/blog/command-a-plus"},{"benchmarkId":"scicode-source-release-snapshot-version-as-labeled","evidenceKind":"lab_self_report","harnessId":"scicode-source-release-snapshot-version-as-labeled:cohere:NjVwcm9ibGVtcy8yODhzdWJwcm9ibGVtczsgc2NpZW50aXN0LWFubm90YXRlZCBiYWNrZ3JvdW5kLg","id":"launch-cohere-1976","ingestRunId":"launch-cohere","modelId":"command-a-reasoning-08-2025","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"65problems/288subproblems; scientist-annotated background.","family":"SciCode","locator":"Image3","metric":"%","notes":"First-party chart numeric label, visually verified.","sourceId":"cohere","sourceTitle":"Command A+ launch benchmarks"},"score":30,"scoreUnit":"percent","sourceUrl":"https://cohere.com/blog/command-a-plus"},{"benchmarkId":"north-agentic-question-answering-source-release-snapshot-version-as-labeled","evidenceKind":"lab_self_report","harnessId":"north-agentic-question-answering-source-release-snapshot-version-as-labeled:cohere:SW50ZXJuYWwgTm9ydGggZW50ZXJwcmlzZSBNQ1AgY2xvdWQtZmlsZSBRQSxMTE0ganVkZ2Uu","id":"launch-cohere-1977","ingestRunId":"launch-cohere","modelId":"command-a-reasoning-08-2025","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Internal North enterprise MCP cloud-file QA,LLM judge.","family":"North Agentic Question Answering","locator":"Image4","metric":"%","notes":"First-party chart numeric label, visually verified.","sourceId":"cohere","sourceTitle":"Command A+ launch benchmarks"},"score":45,"scoreUnit":"percent","sourceUrl":"https://cohere.com/blog/command-a-plus"},{"benchmarkId":"north-data-analysis-source-release-snapshot-version-as-labeled","evidenceKind":"lab_self_report","harnessId":"north-data-analysis-source-release-snapshot-version-as-labeled:cohere:SW50ZXJuYWwgTm9ydGggdXBsb2FkZWQgc3ByZWFkc2hlZXQgZGF0YS1zY2llbmNlIHRhc2tzLExMTSBqdWRnZS4","id":"launch-cohere-1978","ingestRunId":"launch-cohere","modelId":"command-a-reasoning-08-2025","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Internal North uploaded spreadsheet data-science tasks,LLM judge.","family":"North Data Analysis","locator":"Image4","metric":"%","notes":"First-party chart numeric label, visually verified.","sourceId":"cohere","sourceTitle":"Command A+ launch benchmarks"},"score":13,"scoreUnit":"percent","sourceUrl":"https://cohere.com/blog/command-a-plus"},{"benchmarkId":"mt-aime-2025-arabic-japanese-korean-source-release-snapshot-version-as-labeled","evidenceKind":"lab_self_report","harnessId":"mt-aime-2025-arabic-japanese-korean-source-release-snapshot-version-as-labeled:cohere:SW50ZXJuYWwgQ29tbWFuZCBBIFRyYW5zbGF0ZSB0cmFuc2xhdGlvbnM7IEFyYWJpYyxKYXBhbmVzZSxLb3JlYW4u","id":"launch-cohere-1979","ingestRunId":"launch-cohere","modelId":"command-a-reasoning-08-2025","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Internal Command A Translate translations; Arabic,Japanese,Korean.","family":"MT-AIME 2025 Arabic/Japanese/Korean","locator":"Image6","metric":"%","notes":"First-party chart numeric label, visually verified.","sourceId":"cohere","sourceTitle":"Command A+ launch benchmarks"},"score":53,"scoreUnit":"percent","sourceUrl":"https://cohere.com/blog/command-a-plus"},{"benchmarkId":"wmt24-50-varieties-source-release-snapshot-version-as-labeled","evidenceKind":"lab_self_report","harnessId":"wmt24-50-varieties-source-release-snapshot-version-as-labeled:cohere:eENPTUVUeGwgYXZlcmFnZTUwIHZhcmlldGllcyxpbmNsdWRpbmcgaW50ZXJuYWwgSXJpc2gvTWFsdGVzZSB0cmFuc2xhdGlvbnMgYW5kIFNlcmJpYW4gdHJhbnNsaXRlcmF0aW9uLg","id":"launch-cohere-1980","ingestRunId":"launch-cohere","modelId":"command-a-reasoning-08-2025","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"xCOMETxl average50 varieties,including internal Irish/Maltese translations and Serbian transliteration.","family":"WMT24++ 50 varieties","locator":"Image6","metric":"xCOMETxl score","notes":"First-party chart numeric label, visually verified.","sourceId":"cohere","sourceTitle":"Command A+ launch benchmarks"},"score":73,"scoreUnit":"index","sourceUrl":"https://cohere.com/blog/command-a-plus"},{"benchmarkId":"charxiv-descriptive-source-release-snapshot-version-as-labeled","evidenceKind":"lab_self_report","harnessId":"charxiv-descriptive-source-release-snapshot-version-as-labeled:cohere:U3RhbmRhcmQgbWV0aG9kb2xvZ3k7IGludGVnZXIgcm91bmRlZCBsYWJlbHMgaW4gY2hhcnQu","id":"launch-cohere-1981","ingestRunId":"launch-cohere","modelId":"command-a-plus-05-2026","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Standard methodology; integer rounded labels in chart.","family":"CharXiv descriptive","locator":"Image5","metric":"%","notes":"First-party chart numeric label, visually verified.","sourceId":"cohere","sourceTitle":"Command A+ launch benchmarks"},"score":88,"scoreUnit":"percent","sourceUrl":"https://cohere.com/blog/command-a-plus"},{"benchmarkId":"charxiv-descriptive-source-release-snapshot-version-as-labeled","evidenceKind":"lab_self_report","harnessId":"charxiv-descriptive-source-release-snapshot-version-as-labeled:cohere:U3RhbmRhcmQgbWV0aG9kb2xvZ3k7IGludGVnZXIgcm91bmRlZCBsYWJlbHMgaW4gY2hhcnQu","id":"launch-cohere-1982","ingestRunId":"launch-cohere","modelId":"command-a-vision-07-2025","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Standard methodology; integer rounded labels in chart.","family":"CharXiv descriptive","locator":"Image5","metric":"%","notes":"First-party chart numeric label, visually verified.","sourceId":"cohere","sourceTitle":"Command A+ launch benchmarks"},"score":82,"scoreUnit":"percent","sourceUrl":"https://cohere.com/blog/command-a-plus"},{"benchmarkId":"aa-intelligence-index-source-release-snapshot-version-as-labeled","evidenceKind":"lab_self_report","harnessId":"aa-intelligence-index-source-release-snapshot-version-as-labeled:cohere:Q29oZXJlIHF1b3RlZCBBQSBsYXVuY2ggc25hcHNob3Q7IGluZGV4IHZlcnNpb24gdW5zcGVjaWZpZWQu","id":"launch-cohere-1983","ingestRunId":"launch-cohere","modelId":"command-a-plus-05-2026","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Cohere quoted AA launch snapshot; index version unspecified.","family":"AA Intelligence Index","locator":"Article prose","metric":"index points","notes":"First-party chart numeric label, visually verified.","sourceId":"cohere","sourceTitle":"Command A+ launch benchmarks"},"score":37,"scoreUnit":"index","sourceUrl":"https://cohere.com/blog/command-a-plus"},{"benchmarkId":"hle-wo-w-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hle-wo-w-tools-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQuIHdpdGhvdXQgdG9vbHM","id":"launch-deepseek-1136","ingestRunId":"launch-deepseek","modelId":"deepseek-v4-pro","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"hard_reasoning","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card. without tools","family":"Humanity's Last Exam","locator":"Performance table: HLE (wo / w tools)","metric":"%","notes":"First-party reported result.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":42.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"hle-wo-w-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hle-wo-w-tools-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQuIHdpdGggdG9vbHM","id":"launch-deepseek-1137","ingestRunId":"launch-deepseek","modelId":"deepseek-v4-pro","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"hard_reasoning","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card. with tools","family":"Humanity's Last Exam","locator":"Performance table: HLE (wo / w tools)","metric":"%","notes":"First-party reported result.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":60,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"hle-wo-w-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hle-wo-w-tools-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQuIHdpdGhvdXQgdG9vbHM","id":"launch-deepseek-1138","ingestRunId":"launch-deepseek","modelId":"deepseek-v4-flash","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"hard_reasoning","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card. without tools","family":"Humanity's Last Exam","locator":"Performance table: HLE (wo / w tools)","metric":"%","notes":"First-party reported result.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":37.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"hle-wo-w-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hle-wo-w-tools-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQuIHdpdGggdG9vbHM","id":"launch-deepseek-1139","ingestRunId":"launch-deepseek","modelId":"deepseek-v4-flash","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"hard_reasoning","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card. with tools","family":"Humanity's Last Exam","locator":"Performance table: HLE (wo / w tools)","metric":"%","notes":"First-party reported result.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":51.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"hle-wo-w-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hle-wo-w-tools-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQuIHdpdGhvdXQgdG9vbHM","id":"launch-deepseek-1140","ingestRunId":"launch-deepseek","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"hard_reasoning","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card. without tools","family":"Humanity's Last Exam","locator":"Performance table: HLE (wo / w tools)","metric":"%","notes":"First-party reported result.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":40.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"hle-wo-w-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hle-wo-w-tools-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQuIHdpdGggdG9vbHM","id":"launch-deepseek-1141","ingestRunId":"launch-deepseek","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"hard_reasoning","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card. with tools","family":"Humanity's Last Exam","locator":"Performance table: HLE (wo / w tools)","metric":"%","notes":"First-party reported result.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":54.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"hle-wo-w-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hle-wo-w-tools-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQuIHdpdGhvdXQgdG9vbHM","id":"launch-deepseek-1142","ingestRunId":"launch-deepseek","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"hard_reasoning","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card. without tools","family":"Humanity's Last Exam","locator":"Performance table: HLE (wo / w tools)","metric":"%","notes":"First-party reported result.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":43.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"hle-wo-w-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hle-wo-w-tools-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQuIHdpdGggdG9vbHM","id":"launch-deepseek-1143","ingestRunId":"launch-deepseek","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"hard_reasoning","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card. with tools","family":"Humanity's Last Exam","locator":"Performance table: HLE (wo / w tools)","metric":"%","notes":"First-party reported result.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":56,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"hle-wo-w-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hle-wo-w-tools-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQuIHdpdGhvdXQgdG9vbHM","id":"launch-deepseek-1144","ingestRunId":"launch-deepseek","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"hard_reasoning","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card. without tools","family":"Humanity's Last Exam","locator":"Performance table: HLE (wo / w tools)","metric":"%","notes":"First-party reported result.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":49.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"hle-wo-w-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hle-wo-w-tools-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQuIHdpdGggdG9vbHM","id":"launch-deepseek-1145","ingestRunId":"launch-deepseek","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"hard_reasoning","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card. with tools","family":"Humanity's Last Exam","locator":"Performance table: HLE (wo / w tools)","metric":"%","notes":"First-party reported result.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":57.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"hle-wo-w-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hle-wo-w-tools-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQuIHdpdGhvdXQgdG9vbHM","id":"launch-deepseek-1146","ingestRunId":"launch-deepseek","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"hard_reasoning","comparabilityReason":"The source labels this column Fable-5 (w/ fallback); it does not establish standalone Fable 5 performance. Retained as published evidence, excluded from comparable ranking inputs.","comparable":false,"configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card. without tools","family":"Humanity's Last Exam","locator":"Performance table: HLE (wo / w tools)","metric":"%","notes":"First-party reported result.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":53.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"hle-wo-w-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hle-wo-w-tools-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQuIHdpdGggdG9vbHM","id":"launch-deepseek-1147","ingestRunId":"launch-deepseek","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"hard_reasoning","comparabilityReason":"The source labels this column Fable-5 (w/ fallback); it does not establish standalone Fable 5 performance. Retained as published evidence, excluded from comparable ranking inputs.","comparable":false,"configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card. with tools","family":"Humanity's Last Exam","locator":"Performance table: HLE (wo / w tools)","metric":"%","notes":"First-party reported result.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":63,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"terminal-bench-2-1","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","id":"launch-deepseek-1148","ingestRunId":"launch-deepseek","modelId":"deepseek-v4-pro","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"coding","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card.","family":"Terminal-Bench","locator":"Performance table: Terminal Bench 2.1","metric":"%","notes":"First-party reported result.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":87.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"terminal-bench-2-1","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","id":"launch-deepseek-1149","ingestRunId":"launch-deepseek","modelId":"deepseek-v4-flash","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"coding","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card.","family":"Terminal-Bench","locator":"Performance table: Terminal Bench 2.1","metric":"%","notes":"First-party reported result.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":82.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"terminal-bench-2-1","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","id":"launch-deepseek-1150","ingestRunId":"launch-deepseek","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"coding","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card.","family":"Terminal-Bench","locator":"Performance table: Terminal Bench 2.1","metric":"%","notes":"First-party reported result.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":81,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"terminal-bench-2-1","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","id":"launch-deepseek-1151","ingestRunId":"launch-deepseek","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"coding","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card.","family":"Terminal-Bench","locator":"Performance table: Terminal Bench 2.1","metric":"%","notes":"First-party reported result.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":88.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"terminal-bench-2-1","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","id":"launch-deepseek-1152","ingestRunId":"launch-deepseek","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"coding","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card.","family":"Terminal-Bench","locator":"Performance table: Terminal Bench 2.1","metric":"%","notes":"First-party reported result.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":85,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"terminal-bench-2-1","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","id":"launch-deepseek-1153","ingestRunId":"launch-deepseek","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"coding","comparabilityReason":"The source labels this column Fable-5 (w/ fallback); it does not establish standalone Fable 5 performance. Retained as published evidence, excluded from comparable ranking inputs.","comparable":false,"configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card.","family":"Terminal-Bench","locator":"Performance table: Terminal Bench 2.1","metric":"%","notes":"First-party reported result.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":88,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"nl2repo-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"nl2repo-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","id":"launch-deepseek-1154","ingestRunId":"launch-deepseek","modelId":"deepseek-v4-pro","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"coding","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card.","family":"NL2Repo","locator":"Performance table: NL2Repo","metric":"%","notes":"First-party reported result.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":61.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"nl2repo-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"nl2repo-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","id":"launch-deepseek-1155","ingestRunId":"launch-deepseek","modelId":"deepseek-v4-flash","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"coding","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card.","family":"NL2Repo","locator":"Performance table: NL2Repo","metric":"%","notes":"First-party reported result.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":54.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"nl2repo-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"nl2repo-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","id":"launch-deepseek-1156","ingestRunId":"launch-deepseek","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"coding","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card.","family":"NL2Repo","locator":"Performance table: NL2Repo","metric":"%","notes":"First-party reported result.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":48.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"nl2repo-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"nl2repo-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","id":"launch-deepseek-1157","ingestRunId":"launch-deepseek","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"coding","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card.","family":"NL2Repo","locator":"Performance table: NL2Repo","metric":"%","notes":"First-party reported result.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":69.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"cybergym-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"cybergym-source-release-snapshot-version-not-specified-pct:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","id":"launch-deepseek-1158","ingestRunId":"launch-deepseek","modelId":"deepseek-v4-pro","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"coding","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card.","family":"Cybergym","locator":"Performance table: Cybergym","metric":"%","notes":"First-party reported result.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":83.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"cybergym-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"cybergym-source-release-snapshot-version-not-specified-pct:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","id":"launch-deepseek-1159","ingestRunId":"launch-deepseek","modelId":"deepseek-v4-flash","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"coding","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card.","family":"Cybergym","locator":"Performance table: Cybergym","metric":"%","notes":"First-party reported result.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":76.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"cybergym-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"cybergym-source-release-snapshot-version-not-specified-pct:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","id":"launch-deepseek-1160","ingestRunId":"launch-deepseek","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"coding","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card.","family":"Cybergym","locator":"Performance table: Cybergym","metric":"%","notes":"First-party reported result.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":80,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"cybergym-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"cybergym-source-release-snapshot-version-not-specified-pct:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","id":"launch-deepseek-1161","ingestRunId":"launch-deepseek","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"coding","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card.","family":"Cybergym","locator":"Performance table: Cybergym","metric":"%","notes":"First-party reported result.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":78.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"cybergym-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"cybergym-source-release-snapshot-version-not-specified-pct:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","id":"launch-deepseek-1162","ingestRunId":"launch-deepseek","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"coding","comparabilityReason":"The source labels this column Fable-5 (w/ fallback); it does not establish standalone Fable 5 performance. Retained as published evidence, excluded from comparable ranking inputs.","comparable":false,"configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card.","family":"Cybergym","locator":"Performance table: Cybergym","metric":"%","notes":"First-party reported result.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":83.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"deepswe-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"deepswe-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","id":"launch-deepseek-1163","ingestRunId":"launch-deepseek","modelId":"deepseek-v4-pro","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"coding","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card.","family":"DeepSWE","locator":"Performance table: DeepSWE","metric":"%","notes":"First-party reported result.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":62.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"deepswe-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"deepswe-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","id":"launch-deepseek-1164","ingestRunId":"launch-deepseek","modelId":"deepseek-v4-flash","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"coding","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card.","family":"DeepSWE","locator":"Performance table: DeepSWE","metric":"%","notes":"First-party reported result.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":54.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"deepswe-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"deepswe-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","id":"launch-deepseek-1165","ingestRunId":"launch-deepseek","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"coding","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card.","family":"DeepSWE","locator":"Performance table: DeepSWE","metric":"%","notes":"First-party reported result.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":46.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"deepswe-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"deepswe-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","id":"launch-deepseek-1166","ingestRunId":"launch-deepseek","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"coding","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card.","family":"DeepSWE","locator":"Performance table: DeepSWE","metric":"%","notes":"First-party reported result.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":67.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"deepswe-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"deepswe-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","id":"launch-deepseek-1167","ingestRunId":"launch-deepseek","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"coding","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card.","family":"DeepSWE","locator":"Performance table: DeepSWE","metric":"%","notes":"First-party reported result.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":58,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"deepswe-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"deepswe-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","id":"launch-deepseek-1168","ingestRunId":"launch-deepseek","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"coding","comparabilityReason":"The source labels this column Fable-5 (w/ fallback); it does not establish standalone Fable 5 performance. Retained as published evidence, excluded from comparable ranking inputs.","comparable":false,"configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card.","family":"DeepSWE","locator":"Performance table: DeepSWE","metric":"%","notes":"First-party reported result.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":70,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"toolathlon-verified-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"toolathlon-verified-source-release-snapshot-version-not-specified-pct:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","id":"launch-deepseek-1169","ingestRunId":"launch-deepseek","modelId":"deepseek-v4-pro","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"agentic","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card.","family":"Toolathlon-Verified","locator":"Performance table: Toolathlon-Verified","metric":"%","notes":"First-party reported result.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":74.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"toolathlon-verified-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"toolathlon-verified-source-release-snapshot-version-not-specified-pct:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","id":"launch-deepseek-1170","ingestRunId":"launch-deepseek","modelId":"deepseek-v4-flash","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"agentic","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card.","family":"Toolathlon-Verified","locator":"Performance table: Toolathlon-Verified","metric":"%","notes":"First-party reported result.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":70.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"toolathlon-verified-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"toolathlon-verified-source-release-snapshot-version-not-specified-pct:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","id":"launch-deepseek-1171","ingestRunId":"launch-deepseek","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"agentic","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card.","family":"Toolathlon-Verified","locator":"Performance table: Toolathlon-Verified","metric":"%","notes":"First-party reported result.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":59.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"toolathlon-verified-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"toolathlon-verified-source-release-snapshot-version-not-specified-pct:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","id":"launch-deepseek-1172","ingestRunId":"launch-deepseek","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"agentic","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card.","family":"Toolathlon-Verified","locator":"Performance table: Toolathlon-Verified","metric":"%","notes":"First-party reported result.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":76.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"toolathlon-verified-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"toolathlon-verified-source-release-snapshot-version-not-specified-pct:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","id":"launch-deepseek-1173","ingestRunId":"launch-deepseek","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"agentic","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card.","family":"Toolathlon-Verified","locator":"Performance table: Toolathlon-Verified","metric":"%","notes":"First-party reported result.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":76.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"toolathlon-verified-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"toolathlon-verified-source-release-snapshot-version-not-specified-pct:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","id":"launch-deepseek-1174","ingestRunId":"launch-deepseek","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"agentic","comparabilityReason":"The source labels this column Fable-5 (w/ fallback); it does not establish standalone Fable 5 performance. Retained as published evidence, excluded from comparable ranking inputs.","comparable":false,"configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card.","family":"Toolathlon-Verified","locator":"Performance table: Toolathlon-Verified","metric":"%","notes":"First-party reported result.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":77.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"agents-last-exam-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"agents-last-exam-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","id":"launch-deepseek-1175","ingestRunId":"launch-deepseek","modelId":"deepseek-v4-pro","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"agentic","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card.","family":"Agents' Last Exam","locator":"Performance table: Agents' Last Exam","metric":"%","notes":"First-party reported result.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":25.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"agents-last-exam-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"agents-last-exam-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","id":"launch-deepseek-1176","ingestRunId":"launch-deepseek","modelId":"deepseek-v4-flash","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"agentic","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card.","family":"Agents' Last Exam","locator":"Performance table: Agents' Last Exam","metric":"%","notes":"First-party reported result.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":25.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"agents-last-exam-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"agents-last-exam-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","id":"launch-deepseek-1177","ingestRunId":"launch-deepseek","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"agentic","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card.","family":"Agents' Last Exam","locator":"Performance table: Agents' Last Exam","metric":"%","notes":"First-party reported result.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":23.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"agents-last-exam-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"agents-last-exam-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","id":"launch-deepseek-1178","ingestRunId":"launch-deepseek","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"agentic","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card.","family":"Agents' Last Exam","locator":"Performance table: Agents' Last Exam","metric":"%","notes":"First-party reported result.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":27.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"agents-last-exam-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"agents-last-exam-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","id":"launch-deepseek-1179","ingestRunId":"launch-deepseek","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"agentic","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card.","family":"Agents' Last Exam","locator":"Performance table: Agents' Last Exam","metric":"%","notes":"First-party reported result.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":25.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"automationbench-public-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"automationbench-public-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","id":"launch-deepseek-1180","ingestRunId":"launch-deepseek","modelId":"deepseek-v4-pro","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"agentic","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card.","family":"AutomationBench (Public)","locator":"Performance table: AutomationBench (Public)","metric":"%","notes":"First-party reported result.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":31.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"automationbench-public-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"automationbench-public-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","id":"launch-deepseek-1181","ingestRunId":"launch-deepseek","modelId":"deepseek-v4-flash","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"agentic","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card.","family":"AutomationBench (Public)","locator":"Performance table: AutomationBench (Public)","metric":"%","notes":"First-party reported result.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":25.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"automationbench-public-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"automationbench-public-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","id":"launch-deepseek-1182","ingestRunId":"launch-deepseek","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"agentic","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card.","family":"AutomationBench (Public)","locator":"Performance table: AutomationBench (Public)","metric":"%","notes":"First-party reported result.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":12.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"automationbench-public-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"automationbench-public-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","id":"launch-deepseek-1183","ingestRunId":"launch-deepseek","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"agentic","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card.","family":"AutomationBench (Public)","locator":"Performance table: AutomationBench (Public)","metric":"%","notes":"First-party reported result.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":30.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"automationbench-public-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"automationbench-public-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","id":"launch-deepseek-1184","ingestRunId":"launch-deepseek","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"agentic","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card.","family":"AutomationBench (Public)","locator":"Performance table: AutomationBench (Public)","metric":"%","notes":"First-party reported result.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":27.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"automationbench-public-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"automationbench-public-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","id":"launch-deepseek-1185","ingestRunId":"launch-deepseek","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"agentic","comparabilityReason":"The source labels this column Fable-5 (w/ fallback); it does not establish standalone Fable 5 performance. Retained as published evidence, excluded from comparable ranking inputs.","comparable":false,"configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card.","family":"AutomationBench (Public)","locator":"Performance table: AutomationBench (Public)","metric":"%","notes":"First-party reported result.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":29.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"dsbench-fullstack-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"dsbench-fullstack-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","id":"launch-deepseek-1186","ingestRunId":"launch-deepseek","modelId":"deepseek-v4-pro","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"agentic","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card.","family":"DSBench-FullStack †","locator":"Performance table: DSBench-FullStack †","metric":"%","notes":"First-party reported result. † source footnote applies.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":71.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"dsbench-fullstack-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"dsbench-fullstack-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","id":"launch-deepseek-1187","ingestRunId":"launch-deepseek","modelId":"deepseek-v4-flash","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"agentic","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card.","family":"DSBench-FullStack †","locator":"Performance table: DSBench-FullStack †","metric":"%","notes":"First-party reported result. † source footnote applies.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":68.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"dsbench-fullstack-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"dsbench-fullstack-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","id":"launch-deepseek-1188","ingestRunId":"launch-deepseek","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"agentic","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card.","family":"DSBench-FullStack †","locator":"Performance table: DSBench-FullStack †","metric":"%","notes":"First-party reported result. † source footnote applies.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":61.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"dsbench-fullstack-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"dsbench-fullstack-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","id":"launch-deepseek-1189","ingestRunId":"launch-deepseek","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"agentic","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card.","family":"DSBench-FullStack †","locator":"Performance table: DSBench-FullStack †","metric":"%","notes":"First-party reported result. † source footnote applies.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":73.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"dsbench-fullstack-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"dsbench-fullstack-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","id":"launch-deepseek-1190","ingestRunId":"launch-deepseek","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"agentic","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card.","family":"DSBench-FullStack †","locator":"Performance table: DSBench-FullStack †","metric":"%","notes":"First-party reported result. † source footnote applies.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":71.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"dsbench-fullstack-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"dsbench-fullstack-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","id":"launch-deepseek-1191","ingestRunId":"launch-deepseek","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"agentic","comparabilityReason":"The source labels this column Fable-5 (w/ fallback); it does not establish standalone Fable 5 performance. Retained as published evidence, excluded from comparable ranking inputs.","comparable":false,"configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card.","family":"DSBench-FullStack †","locator":"Performance table: DSBench-FullStack †","metric":"%","notes":"First-party reported result. † source footnote applies.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":77.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"dsbench-hard-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"dsbench-hard-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","id":"launch-deepseek-1192","ingestRunId":"launch-deepseek","modelId":"deepseek-v4-pro","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"agentic","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card.","family":"DSBench-Hard †","locator":"Performance table: DSBench-Hard †","metric":"%","notes":"First-party reported result. † source footnote applies.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":67.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"dsbench-hard-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"dsbench-hard-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","id":"launch-deepseek-1193","ingestRunId":"launch-deepseek","modelId":"deepseek-v4-flash","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"agentic","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card.","family":"DSBench-Hard †","locator":"Performance table: DSBench-Hard †","metric":"%","notes":"First-party reported result. † source footnote applies.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":59.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"dsbench-hard-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"dsbench-hard-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","id":"launch-deepseek-1194","ingestRunId":"launch-deepseek","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"agentic","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card.","family":"DSBench-Hard †","locator":"Performance table: DSBench-Hard †","metric":"%","notes":"First-party reported result. † source footnote applies.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":54.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"dsbench-hard-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"dsbench-hard-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","id":"launch-deepseek-1195","ingestRunId":"launch-deepseek","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"agentic","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card.","family":"DSBench-Hard †","locator":"Performance table: DSBench-Hard †","metric":"%","notes":"First-party reported result. † source footnote applies.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":63,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"dsbench-hard-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"dsbench-hard-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","id":"launch-deepseek-1196","ingestRunId":"launch-deepseek","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"agentic","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card.","family":"DSBench-Hard †","locator":"Performance table: DSBench-Hard †","metric":"%","notes":"First-party reported result. † source footnote applies.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":71.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"dsbench-hard-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"dsbench-hard-source-release-snapshot-version-not-specified:deepseek:RGVlcFNlZWsgdXBkYXRlZCAwODEzLzA3MzEgcmVsZWFzZSBldmFsdWF0aW9uIHRhYmxlOyB0b29sL2hhcm5lc3Mgc2V0dGluZ3MgcGVyIG1vZGVsIGNhcmQu","id":"launch-deepseek-1197","ingestRunId":"launch-deepseek","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"61755d88e95789fcd7a36f50892f97bba977a30fc99d0f2907ab787ed10b0e66","report":{"category":"agentic","comparabilityReason":"The source labels this column Fable-5 (w/ fallback); it does not establish standalone Fable 5 performance. Retained as published evidence, excluded from comparable ranking inputs.","comparable":false,"configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card.","family":"DSBench-Hard †","locator":"Performance table: DSBench-Hard †","metric":"%","notes":"First-party reported result. † source footnote applies.","sourceId":"deepseek","sourceTitle":"deepseek-ai/DeepSeek-V4-Pro-0813"},"score":68.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md"},{"benchmarkId":"finance-agent-v2-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"finance-agent-v2-source-release-snapshot-version-not-specified:google:R29vZ2xlIGxhdW5jaCBjaGFydDsgYmVuY2htYXJrIG1ldGhvZG9sb2d5IGxpbmtlZCBvbiBwYWdlOyByZXBvcnRlZCBjb21wYXJhdG9yIHNldHRpbmdzIHZhcnku","id":"launch-google-1357","ingestRunId":"launch-google","modelId":"gemini-3.8-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Google launch chart; benchmark methodology linked on page; reported comparator settings vary.","family":"Finance Agent v2","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"google","sourceTitle":"Gemini 3.8 Flash launch performance"},"score":61.4,"scoreUnit":"percent","sourceUrl":"https://deepmind.google/models/gemini/flash/"},{"benchmarkId":"finance-agent-v2-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"finance-agent-v2-source-release-snapshot-version-not-specified:google:R29vZ2xlIGxhdW5jaCBjaGFydDsgYmVuY2htYXJrIG1ldGhvZG9sb2d5IGxpbmtlZCBvbiBwYWdlOyByZXBvcnRlZCBjb21wYXJhdG9yIHNldHRpbmdzIHZhcnku","id":"launch-google-1358","ingestRunId":"launch-google","modelId":"gemini-3.7-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Google launch chart; benchmark methodology linked on page; reported comparator settings vary.","family":"Finance Agent v2","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"google","sourceTitle":"Gemini 3.8 Flash launch performance"},"score":59,"scoreUnit":"percent","sourceUrl":"https://deepmind.google/models/gemini/flash/"},{"benchmarkId":"finance-agent-v2-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"finance-agent-v2-source-release-snapshot-version-not-specified:google:R29vZ2xlIGxhdW5jaCBjaGFydDsgYmVuY2htYXJrIG1ldGhvZG9sb2d5IGxpbmtlZCBvbiBwYWdlOyByZXBvcnRlZCBjb21wYXJhdG9yIHNldHRpbmdzIHZhcnku","id":"launch-google-1359","ingestRunId":"launch-google","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Google launch chart; benchmark methodology linked on page; reported comparator settings vary.","family":"Finance Agent v2","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"google","sourceTitle":"Gemini 3.8 Flash launch performance"},"score":58.6,"scoreUnit":"percent","sourceUrl":"https://deepmind.google/models/gemini/flash/"},{"benchmarkId":"finance-agent-v2-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"finance-agent-v2-source-release-snapshot-version-not-specified:google:R29vZ2xlIGxhdW5jaCBjaGFydDsgYmVuY2htYXJrIG1ldGhvZG9sb2d5IGxpbmtlZCBvbiBwYWdlOyByZXBvcnRlZCBjb21wYXJhdG9yIHNldHRpbmdzIHZhcnku","id":"launch-google-1360","ingestRunId":"launch-google","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Google launch chart; benchmark methodology linked on page; reported comparator settings vary.","family":"Finance Agent v2","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"google","sourceTitle":"Gemini 3.8 Flash launch performance"},"score":54.4,"scoreUnit":"percent","sourceUrl":"https://deepmind.google/models/gemini/flash/"},{"benchmarkId":"finance-agent-v2-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"finance-agent-v2-source-release-snapshot-version-not-specified:google:R29vZ2xlIGxhdW5jaCBjaGFydDsgYmVuY2htYXJrIG1ldGhvZG9sb2d5IGxpbmtlZCBvbiBwYWdlOyByZXBvcnRlZCBjb21wYXJhdG9yIHNldHRpbmdzIHZhcnku","id":"launch-google-1361","ingestRunId":"launch-google","modelId":"claude-sonnet-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Google launch chart; benchmark methodology linked on page; reported comparator settings vary.","family":"Finance Agent v2","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"google","sourceTitle":"Gemini 3.8 Flash launch performance"},"score":53.9,"scoreUnit":"percent","sourceUrl":"https://deepmind.google/models/gemini/flash/"},{"benchmarkId":"finance-agent-v2-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"finance-agent-v2-source-release-snapshot-version-not-specified:google:R29vZ2xlIGxhdW5jaCBjaGFydDsgYmVuY2htYXJrIG1ldGhvZG9sb2d5IGxpbmtlZCBvbiBwYWdlOyByZXBvcnRlZCBjb21wYXJhdG9yIHNldHRpbmdzIHZhcnku","id":"launch-google-1362","ingestRunId":"launch-google","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Google launch chart; benchmark methodology linked on page; reported comparator settings vary.","family":"Finance Agent v2","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"google","sourceTitle":"Gemini 3.8 Flash launch performance"},"score":53.8,"scoreUnit":"percent","sourceUrl":"https://deepmind.google/models/gemini/flash/"},{"benchmarkId":"harvey-legal-agent-benchmark-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"harvey-legal-agent-benchmark-source-release-snapshot-version-not-specified:google:R29vZ2xlIGxhdW5jaCBjaGFydDsgYmVuY2htYXJrIG1ldGhvZG9sb2d5IGxpbmtlZCBvbiBwYWdlOyByZXBvcnRlZCBjb21wYXJhdG9yIHNldHRpbmdzIHZhcnku","id":"launch-google-1363","ingestRunId":"launch-google","modelId":"gemini-3.8-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Google launch chart; benchmark methodology linked on page; reported comparator settings vary.","family":"Harvey Legal Agent Benchmark","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"google","sourceTitle":"Gemini 3.8 Flash launch performance"},"score":10,"scoreUnit":"percent","sourceUrl":"https://deepmind.google/models/gemini/flash/"},{"benchmarkId":"harvey-legal-agent-benchmark-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"harvey-legal-agent-benchmark-source-release-snapshot-version-not-specified:google:R29vZ2xlIGxhdW5jaCBjaGFydDsgYmVuY2htYXJrIG1ldGhvZG9sb2d5IGxpbmtlZCBvbiBwYWdlOyByZXBvcnRlZCBjb21wYXJhdG9yIHNldHRpbmdzIHZhcnku","id":"launch-google-1364","ingestRunId":"launch-google","modelId":"gemini-3.7-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Google launch chart; benchmark methodology linked on page; reported comparator settings vary.","family":"Harvey Legal Agent Benchmark","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"google","sourceTitle":"Gemini 3.8 Flash launch performance"},"score":8.8,"scoreUnit":"percent","sourceUrl":"https://deepmind.google/models/gemini/flash/"},{"benchmarkId":"harvey-legal-agent-benchmark-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"harvey-legal-agent-benchmark-source-release-snapshot-version-not-specified:google:R29vZ2xlIGxhdW5jaCBjaGFydDsgYmVuY2htYXJrIG1ldGhvZG9sb2d5IGxpbmtlZCBvbiBwYWdlOyByZXBvcnRlZCBjb21wYXJhdG9yIHNldHRpbmdzIHZhcnku","id":"launch-google-1365","ingestRunId":"launch-google","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Google launch chart; benchmark methodology linked on page; reported comparator settings vary.","family":"Harvey Legal Agent Benchmark","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"google","sourceTitle":"Gemini 3.8 Flash launch performance"},"score":6.7,"scoreUnit":"percent","sourceUrl":"https://deepmind.google/models/gemini/flash/"},{"benchmarkId":"harvey-legal-agent-benchmark-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"harvey-legal-agent-benchmark-source-release-snapshot-version-not-specified:google:R29vZ2xlIGxhdW5jaCBjaGFydDsgYmVuY2htYXJrIG1ldGhvZG9sb2d5IGxpbmtlZCBvbiBwYWdlOyByZXBvcnRlZCBjb21wYXJhdG9yIHNldHRpbmdzIHZhcnku","id":"launch-google-1366","ingestRunId":"launch-google","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Google launch chart; benchmark methodology linked on page; reported comparator settings vary.","family":"Harvey Legal Agent Benchmark","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"google","sourceTitle":"Gemini 3.8 Flash launch performance"},"score":0.8,"scoreUnit":"percent","sourceUrl":"https://deepmind.google/models/gemini/flash/"},{"benchmarkId":"harvey-legal-agent-benchmark-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"harvey-legal-agent-benchmark-source-release-snapshot-version-not-specified:google:R29vZ2xlIGxhdW5jaCBjaGFydDsgYmVuY2htYXJrIG1ldGhvZG9sb2d5IGxpbmtlZCBvbiBwYWdlOyByZXBvcnRlZCBjb21wYXJhdG9yIHNldHRpbmdzIHZhcnku","id":"launch-google-1367","ingestRunId":"launch-google","modelId":"claude-sonnet-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Google launch chart; benchmark methodology linked on page; reported comparator settings vary.","family":"Harvey Legal Agent Benchmark","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"google","sourceTitle":"Gemini 3.8 Flash launch performance"},"score":5,"scoreUnit":"percent","sourceUrl":"https://deepmind.google/models/gemini/flash/"},{"benchmarkId":"harvey-legal-agent-benchmark-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"harvey-legal-agent-benchmark-source-release-snapshot-version-not-specified:google:R29vZ2xlIGxhdW5jaCBjaGFydDsgYmVuY2htYXJrIG1ldGhvZG9sb2d5IGxpbmtlZCBvbiBwYWdlOyByZXBvcnRlZCBjb21wYXJhdG9yIHNldHRpbmdzIHZhcnku","id":"launch-google-1368","ingestRunId":"launch-google","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Google launch chart; benchmark methodology linked on page; reported comparator settings vary.","family":"Harvey Legal Agent Benchmark","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"google","sourceTitle":"Gemini 3.8 Flash launch performance"},"score":2.5,"scoreUnit":"percent","sourceUrl":"https://deepmind.google/models/gemini/flash/"},{"benchmarkId":"hle-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hle-verified-source-release-snapshot-version-not-specified:google:R29vZ2xlIGxhdW5jaCBjaGFydDsgYmVuY2htYXJrIG1ldGhvZG9sb2d5IGxpbmtlZCBvbiBwYWdlOyByZXBvcnRlZCBjb21wYXJhdG9yIHNldHRpbmdzIHZhcnku","id":"launch-google-1369","ingestRunId":"launch-google","modelId":"gemini-3.8-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Google launch chart; benchmark methodology linked on page; reported comparator settings vary.","family":"Humanity's Last Exam","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"google","sourceTitle":"Gemini 3.8 Flash launch performance"},"score":54.9,"scoreUnit":"percent","sourceUrl":"https://deepmind.google/models/gemini/flash/"},{"benchmarkId":"hle-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hle-verified-source-release-snapshot-version-not-specified:google:R29vZ2xlIGxhdW5jaCBjaGFydDsgYmVuY2htYXJrIG1ldGhvZG9sb2d5IGxpbmtlZCBvbiBwYWdlOyByZXBvcnRlZCBjb21wYXJhdG9yIHNldHRpbmdzIHZhcnku","id":"launch-google-1370","ingestRunId":"launch-google","modelId":"gemini-3.7-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Google launch chart; benchmark methodology linked on page; reported comparator settings vary.","family":"Humanity's Last Exam","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"google","sourceTitle":"Gemini 3.8 Flash launch performance"},"score":53.6,"scoreUnit":"percent","sourceUrl":"https://deepmind.google/models/gemini/flash/"},{"benchmarkId":"hle-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hle-verified-source-release-snapshot-version-not-specified:google:R29vZ2xlIGxhdW5jaCBjaGFydDsgYmVuY2htYXJrIG1ldGhvZG9sb2d5IGxpbmtlZCBvbiBwYWdlOyByZXBvcnRlZCBjb21wYXJhdG9yIHNldHRpbmdzIHZhcnku","id":"launch-google-1371","ingestRunId":"launch-google","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Google launch chart; benchmark methodology linked on page; reported comparator settings vary.","family":"Humanity's Last Exam","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"google","sourceTitle":"Gemini 3.8 Flash launch performance"},"score":54.4,"scoreUnit":"percent","sourceUrl":"https://deepmind.google/models/gemini/flash/"},{"benchmarkId":"hle-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hle-verified-source-release-snapshot-version-not-specified:google:R29vZ2xlIGxhdW5jaCBjaGFydDsgYmVuY2htYXJrIG1ldGhvZG9sb2d5IGxpbmtlZCBvbiBwYWdlOyByZXBvcnRlZCBjb21wYXJhdG9yIHNldHRpbmdzIHZhcnku","id":"launch-google-1372","ingestRunId":"launch-google","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Google launch chart; benchmark methodology linked on page; reported comparator settings vary.","family":"Humanity's Last Exam","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"google","sourceTitle":"Gemini 3.8 Flash launch performance"},"score":51.1,"scoreUnit":"percent","sourceUrl":"https://deepmind.google/models/gemini/flash/"},{"benchmarkId":"hle-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hle-verified-source-release-snapshot-version-not-specified:google:R29vZ2xlIGxhdW5jaCBjaGFydDsgYmVuY2htYXJrIG1ldGhvZG9sb2d5IGxpbmtlZCBvbiBwYWdlOyByZXBvcnRlZCBjb21wYXJhdG9yIHNldHRpbmdzIHZhcnku","id":"launch-google-1373","ingestRunId":"launch-google","modelId":"claude-sonnet-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Google launch chart; benchmark methodology linked on page; reported comparator settings vary.","family":"Humanity's Last Exam","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"google","sourceTitle":"Gemini 3.8 Flash launch performance"},"score":31,"scoreUnit":"percent","sourceUrl":"https://deepmind.google/models/gemini/flash/"},{"benchmarkId":"hle-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hle-verified-source-release-snapshot-version-not-specified:google:R29vZ2xlIGxhdW5jaCBjaGFydDsgYmVuY2htYXJrIG1ldGhvZG9sb2d5IGxpbmtlZCBvbiBwYWdlOyByZXBvcnRlZCBjb21wYXJhdG9yIHNldHRpbmdzIHZhcnku","id":"launch-google-1374","ingestRunId":"launch-google","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Google launch chart; benchmark methodology linked on page; reported comparator settings vary.","family":"Humanity's Last Exam","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"google","sourceTitle":"Gemini 3.8 Flash launch performance"},"score":54.5,"scoreUnit":"percent","sourceUrl":"https://deepmind.google/models/gemini/flash/"},{"benchmarkId":"deepswe-1-1-percent","evidenceKind":"lab_self_report","harnessId":"deepswe-1-1-percent:google-gemini-3-8-card:RGF0YWN1cnZlIGhpZ2hlc3QgcmVwb3J0ZWQgZWZmb3J0OyBHZW1pbmkzLjhzZWxmY29tcHV0ZWQgbWluaS1zd2UtYWdlbnQgaGlnaHRoaW5raW5n","id":"launch-google-gemini-3-8-card-2040","ingestRunId":"launch-google-gemini-3-8-card","modelId":"gemini-3.8-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Datacurve highest reported effort; Gemini3.8selfcomputed mini-swe-agent highthinking","family":"DeepSWE","locator":"Model card page5 Results table","metric":"percent","notes":"Card Opus74 explicitly corrected as erroneous in linkedmethodology; omitted.","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":73.7,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"deepswe-1-1-percent","evidenceKind":"lab_self_report","harnessId":"deepswe-1-1-percent:google-gemini-3-8-card:RGF0YWN1cnZlIGhpZ2hlc3QgcmVwb3J0ZWQgZWZmb3J0OyBHZW1pbmkzLjhzZWxmY29tcHV0ZWQgbWluaS1zd2UtYWdlbnQgaGlnaHRoaW5raW5n","id":"launch-google-gemini-3-8-card-2041","ingestRunId":"launch-google-gemini-3-8-card","modelId":"gemini-3.7-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Datacurve highest reported effort; Gemini3.8selfcomputed mini-swe-agent highthinking","family":"DeepSWE","locator":"Model card page5 Results table","metric":"percent","notes":"Card Opus74 explicitly corrected as erroneous in linkedmethodology; omitted.","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":65.3,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"deepswe-1-1-percent","evidenceKind":"lab_self_report","harnessId":"deepswe-1-1-percent:google-gemini-3-8-card:RGF0YWN1cnZlIGhpZ2hlc3QgcmVwb3J0ZWQgZWZmb3J0OyBHZW1pbmkzLjhzZWxmY29tcHV0ZWQgbWluaS1zd2UtYWdlbnQgaGlnaHRoaW5raW5n","id":"launch-google-gemini-3-8-card-2042","ingestRunId":"launch-google-gemini-3-8-card","modelId":"claude-sonnet-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Datacurve highest reported effort; Gemini3.8selfcomputed mini-swe-agent highthinking","family":"DeepSWE","locator":"Model card page5 Results table","metric":"percent","notes":"Card Opus74 explicitly corrected as erroneous in linkedmethodology; omitted.","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":53.8,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"deepswe-1-1-percent","evidenceKind":"lab_self_report","harnessId":"deepswe-1-1-percent:google-gemini-3-8-card:RGF0YWN1cnZlIGhpZ2hlc3QgcmVwb3J0ZWQgZWZmb3J0OyBHZW1pbmkzLjhzZWxmY29tcHV0ZWQgbWluaS1zd2UtYWdlbnQgaGlnaHRoaW5raW5n","id":"launch-google-gemini-3-8-card-2043","ingestRunId":"launch-google-gemini-3-8-card","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Datacurve highest reported effort; Gemini3.8selfcomputed mini-swe-agent highthinking","family":"DeepSWE","locator":"Model card page5 Results table","metric":"percent","notes":"Card Opus74 explicitly corrected as erroneous in linkedmethodology; omitted.","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":72.7,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"deepswe-1-1-percent","evidenceKind":"lab_self_report","harnessId":"deepswe-1-1-percent:google-gemini-3-8-card:RGF0YWN1cnZlIGhpZ2hlc3QgcmVwb3J0ZWQgZWZmb3J0OyBHZW1pbmkzLjhzZWxmY29tcHV0ZWQgbWluaS1zd2UtYWdlbnQgaGlnaHRoaW5raW5n","id":"launch-google-gemini-3-8-card-2044","ingestRunId":"launch-google-gemini-3-8-card","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Datacurve highest reported effort; Gemini3.8selfcomputed mini-swe-agent highthinking","family":"DeepSWE","locator":"Model card page5 Results table","metric":"percent","notes":"Card Opus74 explicitly corrected as erroneous in linkedmethodology; omitted.","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":69.6,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"gdpval-aa-2","evidenceKind":"lab_self_report","harnessId":"gdpval-aa-2:google-gemini-3-8-card:QXJ0aWZpY2lhbCBBbmFseXNpcyBwdWJsaWNib2FyZCBzbmFwc2hvdDsgZWZmb3J0IGFzIHJlcG9ydGVk","id":"launch-google-gemini-3-8-card-2045","ingestRunId":"launch-google-gemini-3-8-card","modelId":"gemini-3.8-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"Artificial Analysis publicboard snapshot; effort as reported","family":"GDPval-AA","locator":"Model card page5 Results table","metric":"Elo","notes":"","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":1545,"scoreUnit":"elo","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"gdpval-aa-2","evidenceKind":"lab_self_report","harnessId":"gdpval-aa-2:google-gemini-3-8-card:QXJ0aWZpY2lhbCBBbmFseXNpcyBwdWJsaWNib2FyZCBzbmFwc2hvdDsgZWZmb3J0IGFzIHJlcG9ydGVk","id":"launch-google-gemini-3-8-card-2046","ingestRunId":"launch-google-gemini-3-8-card","modelId":"gemini-3.7-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"Artificial Analysis publicboard snapshot; effort as reported","family":"GDPval-AA","locator":"Model card page5 Results table","metric":"Elo","notes":"","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":1482,"scoreUnit":"elo","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"gdpval-aa-2","evidenceKind":"lab_self_report","harnessId":"gdpval-aa-2:google-gemini-3-8-card:QXJ0aWZpY2lhbCBBbmFseXNpcyBwdWJsaWNib2FyZCBzbmFwc2hvdDsgZWZmb3J0IGFzIHJlcG9ydGVk","id":"launch-google-gemini-3-8-card-2047","ingestRunId":"launch-google-gemini-3-8-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"Artificial Analysis publicboard snapshot; effort as reported","family":"GDPval-AA","locator":"Model card page5 Results table","metric":"Elo","notes":"","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":1824,"scoreUnit":"elo","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"gdpval-aa-2","evidenceKind":"lab_self_report","harnessId":"gdpval-aa-2:google-gemini-3-8-card:QXJ0aWZpY2lhbCBBbmFseXNpcyBwdWJsaWNib2FyZCBzbmFwc2hvdDsgZWZmb3J0IGFzIHJlcG9ydGVk","id":"launch-google-gemini-3-8-card-2048","ingestRunId":"launch-google-gemini-3-8-card","modelId":"claude-sonnet-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"Artificial Analysis publicboard snapshot; effort as reported","family":"GDPval-AA","locator":"Model card page5 Results table","metric":"Elo","notes":"","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":1584,"scoreUnit":"elo","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"gdpval-aa-2","evidenceKind":"lab_self_report","harnessId":"gdpval-aa-2:google-gemini-3-8-card:QXJ0aWZpY2lhbCBBbmFseXNpcyBwdWJsaWNib2FyZCBzbmFwc2hvdDsgZWZmb3J0IGFzIHJlcG9ydGVk","id":"launch-google-gemini-3-8-card-2049","ingestRunId":"launch-google-gemini-3-8-card","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"Artificial Analysis publicboard snapshot; effort as reported","family":"GDPval-AA","locator":"Model card page5 Results table","metric":"Elo","notes":"","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":1710,"scoreUnit":"elo","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"gdpval-aa-2","evidenceKind":"lab_self_report","harnessId":"gdpval-aa-2:google-gemini-3-8-card:QXJ0aWZpY2lhbCBBbmFseXNpcyBwdWJsaWNib2FyZCBzbmFwc2hvdDsgZWZmb3J0IGFzIHJlcG9ydGVk","id":"launch-google-gemini-3-8-card-2050","ingestRunId":"launch-google-gemini-3-8-card","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"Artificial Analysis publicboard snapshot; effort as reported","family":"GDPval-AA","locator":"Model card page5 Results table","metric":"Elo","notes":"","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":1528,"scoreUnit":"elo","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"terminal-bench-2-1-percent-higher","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1-percent-higher:google-gemini-3-8-card:VGVybWludXMyb25seTsgR2VtaW5pIHNlbGZjb21wdXRlZCwgb3RoZXJtb2RlbHMgb2ZmaWNpYWxib2FyZC9BcnRpZmljaWFsQW5hbHlzaXM","id":"launch-google-gemini-3-8-card-2051","ingestRunId":"launch-google-gemini-3-8-card","modelId":"gemini-3.8-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Terminus2only; Gemini selfcomputed, othermodels officialboard/ArtificialAnalysis","family":"Terminal-Bench","locator":"Model card page5 Results table","metric":"percent","notes":"","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":89.4,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"terminal-bench-2-1-percent-higher","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1-percent-higher:google-gemini-3-8-card:VGVybWludXMyb25seTsgR2VtaW5pIHNlbGZjb21wdXRlZCwgb3RoZXJtb2RlbHMgb2ZmaWNpYWxib2FyZC9BcnRpZmljaWFsQW5hbHlzaXM","id":"launch-google-gemini-3-8-card-2052","ingestRunId":"launch-google-gemini-3-8-card","modelId":"gemini-3.7-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Terminus2only; Gemini selfcomputed, othermodels officialboard/ArtificialAnalysis","family":"Terminal-Bench","locator":"Model card page5 Results table","metric":"percent","notes":"","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":85.8,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"terminal-bench-2-1-percent-higher","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1-percent-higher:google-gemini-3-8-card:VGVybWludXMyb25seTsgR2VtaW5pIHNlbGZjb21wdXRlZCwgb3RoZXJtb2RlbHMgb2ZmaWNpYWxib2FyZC9BcnRpZmljaWFsQW5hbHlzaXM","id":"launch-google-gemini-3-8-card-2053","ingestRunId":"launch-google-gemini-3-8-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Terminus2only; Gemini selfcomputed, othermodels officialboard/ArtificialAnalysis","family":"Terminal-Bench","locator":"Model card page5 Results table","metric":"percent","notes":"","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":89.1,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"terminal-bench-2-1-percent-higher","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1-percent-higher:google-gemini-3-8-card:VGVybWludXMyb25seTsgR2VtaW5pIHNlbGZjb21wdXRlZCwgb3RoZXJtb2RlbHMgb2ZmaWNpYWxib2FyZC9BcnRpZmljaWFsQW5hbHlzaXM","id":"launch-google-gemini-3-8-card-2054","ingestRunId":"launch-google-gemini-3-8-card","modelId":"claude-sonnet-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Terminus2only; Gemini selfcomputed, othermodels officialboard/ArtificialAnalysis","family":"Terminal-Bench","locator":"Model card page5 Results table","metric":"percent","notes":"","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":80.4,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"terminal-bench-2-1-percent-higher","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1-percent-higher:google-gemini-3-8-card:VGVybWludXMyb25seTsgR2VtaW5pIHNlbGZjb21wdXRlZCwgb3RoZXJtb2RlbHMgb2ZmaWNpYWxib2FyZC9BcnRpZmljaWFsQW5hbHlzaXM","id":"launch-google-gemini-3-8-card-2055","ingestRunId":"launch-google-gemini-3-8-card","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Terminus2only; Gemini selfcomputed, othermodels officialboard/ArtificialAnalysis","family":"Terminal-Bench","locator":"Model card page5 Results table","metric":"percent","notes":"","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":88.8,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"terminal-bench-2-1-percent-higher","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1-percent-higher:google-gemini-3-8-card:VGVybWludXMyb25seTsgR2VtaW5pIHNlbGZjb21wdXRlZCwgb3RoZXJtb2RlbHMgb2ZmaWNpYWxib2FyZC9BcnRpZmljaWFsQW5hbHlzaXM","id":"launch-google-gemini-3-8-card-2056","ingestRunId":"launch-google-gemini-3-8-card","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Terminus2only; Gemini selfcomputed, othermodels officialboard/ArtificialAnalysis","family":"Terminal-Bench","locator":"Model card page5 Results table","metric":"percent","notes":"","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":87.4,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"terminal-bench-4-0-percent","evidenceKind":"lab_self_report","harnessId":"terminal-bench-4-0-percent:google-gemini-3-8-card:T2ZmaWNpYWxwdWJsaWNib2FyZCBoaWdoZXN0IHNjb3JpbmcgdGhpbmtpbmcgbGV2ZWw7IG5hdGl2ZWFnZW50cyBtYXkgZGlmZmVy","id":"launch-google-gemini-3-8-card-2057","ingestRunId":"launch-google-gemini-3-8-card","modelId":"gemini-3.8-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Officialpublicboard highest scoring thinking level; nativeagents may differ","family":"Terminal-Bench","locator":"Model card page5 Results table","metric":"percent","notes":"","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":19.1,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"terminal-bench-4-0-percent","evidenceKind":"lab_self_report","harnessId":"terminal-bench-4-0-percent:google-gemini-3-8-card:T2ZmaWNpYWxwdWJsaWNib2FyZCBoaWdoZXN0IHNjb3JpbmcgdGhpbmtpbmcgbGV2ZWw7IG5hdGl2ZWFnZW50cyBtYXkgZGlmZmVy","id":"launch-google-gemini-3-8-card-2058","ingestRunId":"launch-google-gemini-3-8-card","modelId":"gemini-3.7-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Officialpublicboard highest scoring thinking level; nativeagents may differ","family":"Terminal-Bench","locator":"Model card page5 Results table","metric":"percent","notes":"","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":11.2,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"terminal-bench-4-0-percent","evidenceKind":"lab_self_report","harnessId":"terminal-bench-4-0-percent:google-gemini-3-8-card:T2ZmaWNpYWxwdWJsaWNib2FyZCBoaWdoZXN0IHNjb3JpbmcgdGhpbmtpbmcgbGV2ZWw7IG5hdGl2ZWFnZW50cyBtYXkgZGlmZmVy","id":"launch-google-gemini-3-8-card-2059","ingestRunId":"launch-google-gemini-3-8-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Officialpublicboard highest scoring thinking level; nativeagents may differ","family":"Terminal-Bench","locator":"Model card page5 Results table","metric":"percent","notes":"","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":51.8,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"terminal-bench-4-0-percent","evidenceKind":"lab_self_report","harnessId":"terminal-bench-4-0-percent:google-gemini-3-8-card:T2ZmaWNpYWxwdWJsaWNib2FyZCBoaWdoZXN0IHNjb3JpbmcgdGhpbmtpbmcgbGV2ZWw7IG5hdGl2ZWFnZW50cyBtYXkgZGlmZmVy","id":"launch-google-gemini-3-8-card-2060","ingestRunId":"launch-google-gemini-3-8-card","modelId":"claude-sonnet-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Officialpublicboard highest scoring thinking level; nativeagents may differ","family":"Terminal-Bench","locator":"Model card page5 Results table","metric":"percent","notes":"","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":12.4,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"terminal-bench-4-0-percent","evidenceKind":"lab_self_report","harnessId":"terminal-bench-4-0-percent:google-gemini-3-8-card:T2ZmaWNpYWxwdWJsaWNib2FyZCBoaWdoZXN0IHNjb3JpbmcgdGhpbmtpbmcgbGV2ZWw7IG5hdGl2ZWFnZW50cyBtYXkgZGlmZmVy","id":"launch-google-gemini-3-8-card-2061","ingestRunId":"launch-google-gemini-3-8-card","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Officialpublicboard highest scoring thinking level; nativeagents may differ","family":"Terminal-Bench","locator":"Model card page5 Results table","metric":"percent","notes":"","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":37.3,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"terminal-bench-4-0-percent","evidenceKind":"lab_self_report","harnessId":"terminal-bench-4-0-percent:google-gemini-3-8-card:T2ZmaWNpYWxwdWJsaWNib2FyZCBoaWdoZXN0IHNjb3JpbmcgdGhpbmtpbmcgbGV2ZWw7IG5hdGl2ZWFnZW50cyBtYXkgZGlmZmVy","id":"launch-google-gemini-3-8-card-2062","ingestRunId":"launch-google-gemini-3-8-card","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Officialpublicboard highest scoring thinking level; nativeagents may differ","family":"Terminal-Bench","locator":"Model card page5 Results table","metric":"percent","notes":"","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":23.6,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"gdp-pdf","evidenceKind":"lab_self_report","harnessId":"gdp-pdf:google-gemini-3-8-card:QWxsLXBhc3MgcmF0ZTsgYWxsbW9kZWxzIHNlbGZjb21wdXRlZCBieUdvb2dsZQ","id":"launch-google-gemini-3-8-card-2063","ingestRunId":"launch-google-gemini-3-8-card","modelId":"gemini-3.8-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"All-pass rate; allmodels selfcomputed byGoogle","family":"GDP.PDF","locator":"Model card page5 Results table","metric":"percent","notes":"","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":35,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"gdp-pdf","evidenceKind":"lab_self_report","harnessId":"gdp-pdf:google-gemini-3-8-card:QWxsLXBhc3MgcmF0ZTsgYWxsbW9kZWxzIHNlbGZjb21wdXRlZCBieUdvb2dsZQ","id":"launch-google-gemini-3-8-card-2064","ingestRunId":"launch-google-gemini-3-8-card","modelId":"gemini-3.7-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"All-pass rate; allmodels selfcomputed byGoogle","family":"GDP.PDF","locator":"Model card page5 Results table","metric":"percent","notes":"","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":34,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"gdp-pdf","evidenceKind":"lab_self_report","harnessId":"gdp-pdf:google-gemini-3-8-card:QWxsLXBhc3MgcmF0ZTsgYWxsbW9kZWxzIHNlbGZjb21wdXRlZCBieUdvb2dsZQ","id":"launch-google-gemini-3-8-card-2065","ingestRunId":"launch-google-gemini-3-8-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"All-pass rate; allmodels selfcomputed byGoogle","family":"GDP.PDF","locator":"Model card page5 Results table","metric":"percent","notes":"","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":37,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"gdp-pdf","evidenceKind":"lab_self_report","harnessId":"gdp-pdf:google-gemini-3-8-card:QWxsLXBhc3MgcmF0ZTsgYWxsbW9kZWxzIHNlbGZjb21wdXRlZCBieUdvb2dsZQ","id":"launch-google-gemini-3-8-card-2066","ingestRunId":"launch-google-gemini-3-8-card","modelId":"claude-sonnet-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"All-pass rate; allmodels selfcomputed byGoogle","family":"GDP.PDF","locator":"Model card page5 Results table","metric":"percent","notes":"","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":28,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"gdp-pdf","evidenceKind":"lab_self_report","harnessId":"gdp-pdf:google-gemini-3-8-card:QWxsLXBhc3MgcmF0ZTsgYWxsbW9kZWxzIHNlbGZjb21wdXRlZCBieUdvb2dsZQ","id":"launch-google-gemini-3-8-card-2067","ingestRunId":"launch-google-gemini-3-8-card","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"All-pass rate; allmodels selfcomputed byGoogle","family":"GDP.PDF","locator":"Model card page5 Results table","metric":"percent","notes":"","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":40,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"gdp-pdf","evidenceKind":"lab_self_report","harnessId":"gdp-pdf:google-gemini-3-8-card:QWxsLXBhc3MgcmF0ZTsgYWxsbW9kZWxzIHNlbGZjb21wdXRlZCBieUdvb2dsZQ","id":"launch-google-gemini-3-8-card-2068","ingestRunId":"launch-google-gemini-3-8-card","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"All-pass rate; allmodels selfcomputed byGoogle","family":"GDP.PDF","locator":"Model card page5 Results table","metric":"percent","notes":"","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":29,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"charxiv-reasoning","evidenceKind":"lab_self_report","harnessId":"charxiv-reasoning:google-gemini-3-8-card:Tm8gdG9vbHM7IEdlbWluaS9HUFQvT3B1cyBzZWxmY29tcHV0ZWQ7IFNvbm5ldCBzZWxmcmVwb3J0ZWQ","id":"launch-google-gemini-3-8-card-2069","ingestRunId":"launch-google-gemini-3-8-card","modelId":"gemini-3.8-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"No tools; Gemini/GPT/Opus selfcomputed; Sonnet selfreported","family":"CharXiv Reasoning","locator":"Model card page5 Results table","metric":"percent","notes":"","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":86.2,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"charxiv-reasoning","evidenceKind":"lab_self_report","harnessId":"charxiv-reasoning:google-gemini-3-8-card:Tm8gdG9vbHM7IEdlbWluaS9HUFQvT3B1cyBzZWxmY29tcHV0ZWQ7IFNvbm5ldCBzZWxmcmVwb3J0ZWQ","id":"launch-google-gemini-3-8-card-2070","ingestRunId":"launch-google-gemini-3-8-card","modelId":"gemini-3.7-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"No tools; Gemini/GPT/Opus selfcomputed; Sonnet selfreported","family":"CharXiv Reasoning","locator":"Model card page5 Results table","metric":"percent","notes":"","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":84.5,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"charxiv-reasoning","evidenceKind":"lab_self_report","harnessId":"charxiv-reasoning:google-gemini-3-8-card:Tm8gdG9vbHM7IEdlbWluaS9HUFQvT3B1cyBzZWxmY29tcHV0ZWQ7IFNvbm5ldCBzZWxmcmVwb3J0ZWQ","id":"launch-google-gemini-3-8-card-2071","ingestRunId":"launch-google-gemini-3-8-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"No tools; Gemini/GPT/Opus selfcomputed; Sonnet selfreported","family":"CharXiv Reasoning","locator":"Model card page5 Results table","metric":"percent","notes":"","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":83.7,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"charxiv-reasoning","evidenceKind":"lab_self_report","harnessId":"charxiv-reasoning:google-gemini-3-8-card:Tm8gdG9vbHM7IEdlbWluaS9HUFQvT3B1cyBzZWxmY29tcHV0ZWQ7IFNvbm5ldCBzZWxmcmVwb3J0ZWQ","id":"launch-google-gemini-3-8-card-2072","ingestRunId":"launch-google-gemini-3-8-card","modelId":"claude-sonnet-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"No tools; Gemini/GPT/Opus selfcomputed; Sonnet selfreported","family":"CharXiv Reasoning","locator":"Model card page5 Results table","metric":"percent","notes":"","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":70.1,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"charxiv-reasoning","evidenceKind":"lab_self_report","harnessId":"charxiv-reasoning:google-gemini-3-8-card:Tm8gdG9vbHM7IEdlbWluaS9HUFQvT3B1cyBzZWxmY29tcHV0ZWQ7IFNvbm5ldCBzZWxmcmVwb3J0ZWQ","id":"launch-google-gemini-3-8-card-2073","ingestRunId":"launch-google-gemini-3-8-card","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"No tools; Gemini/GPT/Opus selfcomputed; Sonnet selfreported","family":"CharXiv Reasoning","locator":"Model card page5 Results table","metric":"percent","notes":"","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":85.8,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"charxiv-reasoning","evidenceKind":"lab_self_report","harnessId":"charxiv-reasoning:google-gemini-3-8-card:Tm8gdG9vbHM7IEdlbWluaS9HUFQvT3B1cyBzZWxmY29tcHV0ZWQ7IFNvbm5ldCBzZWxmcmVwb3J0ZWQ","id":"launch-google-gemini-3-8-card-2074","ingestRunId":"launch-google-gemini-3-8-card","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"No tools; Gemini/GPT/Opus selfcomputed; Sonnet selfreported","family":"CharXiv Reasoning","locator":"Model card page5 Results table","metric":"percent","notes":"","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":85.9,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"lvbench-static","evidenceKind":"lab_self_report","harnessId":"lvbench-static:google-gemini-3-8-card:Tm8gdG9vbHM7MTAyNGZyYW1lcyBHZW1pbmkvR1BULDMwMGZyYW1lcyBDbGF1ZGUgZHVlQVBJbGltaXQ7IG1vZGVsLXNwZWNpZmljIGZyYW1lIGJ1ZGdldDogMTAyNA","id":"launch-google-gemini-3-8-card-2075","ingestRunId":"launch-google-gemini-3-8-card","modelId":"gemini-3.8-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"No tools;1024frames Gemini/GPT,300frames Claude dueAPIlimit; model-specific frame budget: 1024","family":"LVBench","locator":"Model card page5 Results table","metric":"percent","notes":"Frame budgets differ; table labels Gemini3.8static explicitly.","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":87.1,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"lvbench-static","evidenceKind":"lab_self_report","harnessId":"lvbench-static:google-gemini-3-8-card:Tm8gdG9vbHM7MTAyNGZyYW1lcyBHZW1pbmkvR1BULDMwMGZyYW1lcyBDbGF1ZGUgZHVlQVBJbGltaXQ7IG1vZGVsLXNwZWNpZmljIGZyYW1lIGJ1ZGdldDogMTAyNA","id":"launch-google-gemini-3-8-card-2076","ingestRunId":"launch-google-gemini-3-8-card","modelId":"gemini-3.7-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"No tools;1024frames Gemini/GPT,300frames Claude dueAPIlimit; model-specific frame budget: 1024","family":"LVBench","locator":"Model card page5 Results table","metric":"percent","notes":"Frame budgets differ; table labels Gemini3.8static explicitly.","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":85.4,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"lvbench-static","evidenceKind":"lab_self_report","harnessId":"lvbench-static:google-gemini-3-8-card:Tm8gdG9vbHM7MTAyNGZyYW1lcyBHZW1pbmkvR1BULDMwMGZyYW1lcyBDbGF1ZGUgZHVlQVBJbGltaXQ7IG1vZGVsLXNwZWNpZmljIGZyYW1lIGJ1ZGdldDogMzAw","id":"launch-google-gemini-3-8-card-2077","ingestRunId":"launch-google-gemini-3-8-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"No tools;1024frames Gemini/GPT,300frames Claude dueAPIlimit; model-specific frame budget: 300","family":"LVBench","locator":"Model card page5 Results table","metric":"percent","notes":"Frame budgets differ; table labels Gemini3.8static explicitly.","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":75.4,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"lvbench-static","evidenceKind":"lab_self_report","harnessId":"lvbench-static:google-gemini-3-8-card:Tm8gdG9vbHM7MTAyNGZyYW1lcyBHZW1pbmkvR1BULDMwMGZyYW1lcyBDbGF1ZGUgZHVlQVBJbGltaXQ7IG1vZGVsLXNwZWNpZmljIGZyYW1lIGJ1ZGdldDogMzAw","id":"launch-google-gemini-3-8-card-2078","ingestRunId":"launch-google-gemini-3-8-card","modelId":"claude-sonnet-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"No tools;1024frames Gemini/GPT,300frames Claude dueAPIlimit; model-specific frame budget: 300","family":"LVBench","locator":"Model card page5 Results table","metric":"percent","notes":"Frame budgets differ; table labels Gemini3.8static explicitly.","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":68.5,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"lvbench-static","evidenceKind":"lab_self_report","harnessId":"lvbench-static:google-gemini-3-8-card:Tm8gdG9vbHM7MTAyNGZyYW1lcyBHZW1pbmkvR1BULDMwMGZyYW1lcyBDbGF1ZGUgZHVlQVBJbGltaXQ7IG1vZGVsLXNwZWNpZmljIGZyYW1lIGJ1ZGdldDogMTAyNA","id":"launch-google-gemini-3-8-card-2079","ingestRunId":"launch-google-gemini-3-8-card","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"No tools;1024frames Gemini/GPT,300frames Claude dueAPIlimit; model-specific frame budget: 1024","family":"LVBench","locator":"Model card page5 Results table","metric":"percent","notes":"Frame budgets differ; table labels Gemini3.8static explicitly.","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":82.1,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"lvbench-static","evidenceKind":"lab_self_report","harnessId":"lvbench-static:google-gemini-3-8-card:Tm8gdG9vbHM7MTAyNGZyYW1lcyBHZW1pbmkvR1BULDMwMGZyYW1lcyBDbGF1ZGUgZHVlQVBJbGltaXQ7IG1vZGVsLXNwZWNpZmljIGZyYW1lIGJ1ZGdldDogMTAyNA","id":"launch-google-gemini-3-8-card-2080","ingestRunId":"launch-google-gemini-3-8-card","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"No tools;1024frames Gemini/GPT,300frames Claude dueAPIlimit; model-specific frame budget: 1024","family":"LVBench","locator":"Model card page5 Results table","metric":"percent","notes":"Frame budgets differ; table labels Gemini3.8static explicitly.","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":78.9,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"lvbench-agentic","evidenceKind":"lab_self_report","harnessId":"lvbench-agentic:google-gemini-3-8-card:Q2FyZCBsYWJlbHMgYWdlbnRpYzsgbGlua2VkbWV0aG9kb2xvZ3kgZGVzY3JpYmVzIG9ubHkgbm8tdG9vbHMgc3RhdGljIHNldHVw","id":"launch-google-gemini-3-8-card-2081","ingestRunId":"launch-google-gemini-3-8-card","modelId":"gemini-3.8-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","comparabilityReason":"Card reports agentic result but linked methodology describes only static no-tools protocol.","comparable":false,"configuration":"Card labels agentic; linkedmethodology describes only no-tools static setup","family":"LVBench","locator":"Model card page5 Results table","metric":"percent","notes":"Agentic tool/protocol details unresolved; rawreportedresult only.","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":87.8,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"osworld-partial-2-0-pre-08-08","evidenceKind":"lab_self_report","harnessId":"osworld-partial-2-0-pre-08-08:google-gemini-3-8-card:UGFydGlhbHNjb3JlOyBiYXRjaHRvb2xzOzEwODBwLzUwMHN0ZXBzOyBHZW1pbmkvU29ubmV0IGJlc3RvZjNydW5zOyBzY3JlZW5zaG90b25seTsgb2ZmaWNpYWxDVUFoYXJuZXNz","id":"launch-google-gemini-3-8-card-2082","ingestRunId":"launch-google-gemini-3-8-card","modelId":"gemini-3.8-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Partialscore; batchtools;1080p/500steps; Gemini/Sonnet bestof3runs; screenshotonly; officialCUAharness","family":"OSWorld","locator":"Model card page5 Results table","metric":"percent","notes":"Methodology says runs pre08.08patch but Opusvalue fromFable5.1card usesAugustfixedtasks; no controlledsameversionclaim. GPT values providerreports.","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":59,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"osworld-partial-2-0-pre-08-08","evidenceKind":"lab_self_report","harnessId":"osworld-partial-2-0-pre-08-08:google-gemini-3-8-card:UGFydGlhbHNjb3JlOyBiYXRjaHRvb2xzOzEwODBwLzUwMHN0ZXBzOyBHZW1pbmkvU29ubmV0IGJlc3RvZjNydW5zOyBzY3JlZW5zaG90b25seTsgb2ZmaWNpYWxDVUFoYXJuZXNz","id":"launch-google-gemini-3-8-card-2083","ingestRunId":"launch-google-gemini-3-8-card","modelId":"gemini-3.7-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Partialscore; batchtools;1080p/500steps; Gemini/Sonnet bestof3runs; screenshotonly; officialCUAharness","family":"OSWorld","locator":"Model card page5 Results table","metric":"percent","notes":"Methodology says runs pre08.08patch but Opusvalue fromFable5.1card usesAugustfixedtasks; no controlledsameversionclaim. GPT values providerreports.","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":50.6,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"osworld-partial-2-0-august2026-fixed-tasks","evidenceKind":"lab_self_report","harnessId":"osworld-partial-2-0-august2026-fixed-tasks:google-gemini-3-8-card:UGFydGlhbHNjb3JlOyBiYXRjaHRvb2xzOzEwODBwLzUwMHN0ZXBzOyBHZW1pbmkvU29ubmV0IGJlc3RvZjNydW5zOyBzY3JlZW5zaG90b25seTsgb2ZmaWNpYWxDVUFoYXJuZXNz","id":"launch-google-gemini-3-8-card-2084","ingestRunId":"launch-google-gemini-3-8-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Partialscore; batchtools;1080p/500steps; Gemini/Sonnet bestof3runs; screenshotonly; officialCUAharness","family":"OSWorld","locator":"Model card page5 Results table","metric":"percent","notes":"Methodology says runs pre08.08patch but Opusvalue fromFable5.1card usesAugustfixedtasks; no controlledsameversionclaim. GPT values providerreports.","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":75.4,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"osworld-partial-2-0-pre-08-08","evidenceKind":"lab_self_report","harnessId":"osworld-partial-2-0-pre-08-08:google-gemini-3-8-card:UGFydGlhbHNjb3JlOyBiYXRjaHRvb2xzOzEwODBwLzUwMHN0ZXBzOyBHZW1pbmkvU29ubmV0IGJlc3RvZjNydW5zOyBzY3JlZW5zaG90b25seTsgb2ZmaWNpYWxDVUFoYXJuZXNz","id":"launch-google-gemini-3-8-card-2085","ingestRunId":"launch-google-gemini-3-8-card","modelId":"claude-sonnet-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Partialscore; batchtools;1080p/500steps; Gemini/Sonnet bestof3runs; screenshotonly; officialCUAharness","family":"OSWorld","locator":"Model card page5 Results table","metric":"percent","notes":"Methodology says runs pre08.08patch but Opusvalue fromFable5.1card usesAugustfixedtasks; no controlledsameversionclaim. GPT values providerreports.","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":42.6,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"osworld-partial-2-0-task-revision-unverified-gpt-5-6-sol","evidenceKind":"lab_self_report","harnessId":"osworld-partial-2-0-task-revision-unverified-gpt-5-6-sol:google-gemini-3-8-card:UGFydGlhbHNjb3JlOyBiYXRjaHRvb2xzOzEwODBwLzUwMHN0ZXBzOyBHZW1pbmkvU29ubmV0IGJlc3RvZjNydW5zOyBzY3JlZW5zaG90b25seTsgb2ZmaWNpYWxDVUFoYXJuZXNz","id":"launch-google-gemini-3-8-card-2086","ingestRunId":"launch-google-gemini-3-8-card","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","comparabilityReason":"Provider-sourced OSWorld task revision is unverified; cannot join a known-version comparison.","comparable":false,"configuration":"Partialscore; batchtools;1080p/500steps; Gemini/Sonnet bestof3runs; screenshotonly; officialCUAharness","family":"OSWorld","locator":"Model card page5 Results table","metric":"percent","notes":"Methodology says runs pre08.08patch but Opusvalue fromFable5.1card usesAugustfixedtasks; no controlledsameversionclaim. GPT values providerreports.","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":62.6,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"osworld-partial-2-0-task-revision-unverified-gpt-5-6-terra","evidenceKind":"lab_self_report","harnessId":"osworld-partial-2-0-task-revision-unverified-gpt-5-6-terra:google-gemini-3-8-card:UGFydGlhbHNjb3JlOyBiYXRjaHRvb2xzOzEwODBwLzUwMHN0ZXBzOyBHZW1pbmkvU29ubmV0IGJlc3RvZjNydW5zOyBzY3JlZW5zaG90b25seTsgb2ZmaWNpYWxDVUFoYXJuZXNz","id":"launch-google-gemini-3-8-card-2087","ingestRunId":"launch-google-gemini-3-8-card","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","comparabilityReason":"Provider-sourced OSWorld task revision is unverified; cannot join a known-version comparison.","comparable":false,"configuration":"Partialscore; batchtools;1080p/500steps; Gemini/Sonnet bestof3runs; screenshotonly; officialCUAharness","family":"OSWorld","locator":"Model card page5 Results table","metric":"percent","notes":"Methodology says runs pre08.08patch but Opusvalue fromFable5.1card usesAugustfixedtasks; no controlledsameversionclaim. GPT values providerreports.","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":50.2,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"biomysterybench-human-solvable","evidenceKind":"lab_self_report","harnessId":"biomysterybench-human-solvable:google-gemini-3-8-card:TGludXh0ZXJtaW5hbCxiaW9pbmZvdG9vbHMsUHl0aG9uLFI7IGFsbG93bGlzdGVkbmV0d29yazsgR2VtaW5pL0dQVHNlbGZjb21wdXRlZCxDbGF1ZGVwcm92aWRlcnJlcG9ydHM","id":"launch-google-gemini-3-8-card-2088","ingestRunId":"launch-google-gemini-3-8-card","modelId":"gemini-3.8-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Linuxterminal,bioinfotools,Python,R; allowlistednetwork; Gemini/GPTselfcomputed,Claudeproviderreports","family":"BioMysteryBench","locator":"Model card page5 Results table","metric":"percent","notes":"","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":88.8,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"biomysterybench-human-solvable","evidenceKind":"lab_self_report","harnessId":"biomysterybench-human-solvable:google-gemini-3-8-card:TGludXh0ZXJtaW5hbCxiaW9pbmZvdG9vbHMsUHl0aG9uLFI7IGFsbG93bGlzdGVkbmV0d29yazsgR2VtaW5pL0dQVHNlbGZjb21wdXRlZCxDbGF1ZGVwcm92aWRlcnJlcG9ydHM","id":"launch-google-gemini-3-8-card-2089","ingestRunId":"launch-google-gemini-3-8-card","modelId":"gemini-3.7-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Linuxterminal,bioinfotools,Python,R; allowlistednetwork; Gemini/GPTselfcomputed,Claudeproviderreports","family":"BioMysteryBench","locator":"Model card page5 Results table","metric":"percent","notes":"","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":87.1,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"biomysterybench-human-solvable","evidenceKind":"lab_self_report","harnessId":"biomysterybench-human-solvable:google-gemini-3-8-card:TGludXh0ZXJtaW5hbCxiaW9pbmZvdG9vbHMsUHl0aG9uLFI7IGFsbG93bGlzdGVkbmV0d29yazsgR2VtaW5pL0dQVHNlbGZjb21wdXRlZCxDbGF1ZGVwcm92aWRlcnJlcG9ydHM","id":"launch-google-gemini-3-8-card-2090","ingestRunId":"launch-google-gemini-3-8-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Linuxterminal,bioinfotools,Python,R; allowlistednetwork; Gemini/GPTselfcomputed,Claudeproviderreports","family":"BioMysteryBench","locator":"Model card page5 Results table","metric":"percent","notes":"","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":90.1,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"biomysterybench-human-solvable","evidenceKind":"lab_self_report","harnessId":"biomysterybench-human-solvable:google-gemini-3-8-card:TGludXh0ZXJtaW5hbCxiaW9pbmZvdG9vbHMsUHl0aG9uLFI7IGFsbG93bGlzdGVkbmV0d29yazsgR2VtaW5pL0dQVHNlbGZjb21wdXRlZCxDbGF1ZGVwcm92aWRlcnJlcG9ydHM","id":"launch-google-gemini-3-8-card-2091","ingestRunId":"launch-google-gemini-3-8-card","modelId":"claude-sonnet-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Linuxterminal,bioinfotools,Python,R; allowlistednetwork; Gemini/GPTselfcomputed,Claudeproviderreports","family":"BioMysteryBench","locator":"Model card page5 Results table","metric":"percent","notes":"","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":87.5,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"biomysterybench-human-solvable","evidenceKind":"lab_self_report","harnessId":"biomysterybench-human-solvable:google-gemini-3-8-card:TGludXh0ZXJtaW5hbCxiaW9pbmZvdG9vbHMsUHl0aG9uLFI7IGFsbG93bGlzdGVkbmV0d29yazsgR2VtaW5pL0dQVHNlbGZjb21wdXRlZCxDbGF1ZGVwcm92aWRlcnJlcG9ydHM","id":"launch-google-gemini-3-8-card-2092","ingestRunId":"launch-google-gemini-3-8-card","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Linuxterminal,bioinfotools,Python,R; allowlistednetwork; Gemini/GPTselfcomputed,Claudeproviderreports","family":"BioMysteryBench","locator":"Model card page5 Results table","metric":"percent","notes":"","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":79.5,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"biomysterybench-human-solvable","evidenceKind":"lab_self_report","harnessId":"biomysterybench-human-solvable:google-gemini-3-8-card:TGludXh0ZXJtaW5hbCxiaW9pbmZvdG9vbHMsUHl0aG9uLFI7IGFsbG93bGlzdGVkbmV0d29yazsgR2VtaW5pL0dQVHNlbGZjb21wdXRlZCxDbGF1ZGVwcm92aWRlcnJlcG9ydHM","id":"launch-google-gemini-3-8-card-2093","ingestRunId":"launch-google-gemini-3-8-card","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Linuxterminal,bioinfotools,Python,R; allowlistednetwork; Gemini/GPTselfcomputed,Claudeproviderreports","family":"BioMysteryBench","locator":"Model card page5 Results table","metric":"percent","notes":"","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":83.8,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"biomysterybench-human-difficult","evidenceKind":"lab_self_report","harnessId":"biomysterybench-human-difficult:google-gemini-3-8-card:TGludXh0ZXJtaW5hbCxiaW9pbmZvdG9vbHMsUHl0aG9uLFI7IGFsbG93bGlzdGVkbmV0d29yazsgR2VtaW5pL0dQVHNlbGZjb21wdXRlZCxDbGF1ZGVwcm92aWRlcnJlcG9ydHM","id":"launch-google-gemini-3-8-card-2094","ingestRunId":"launch-google-gemini-3-8-card","modelId":"gemini-3.8-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Linuxterminal,bioinfotools,Python,R; allowlistednetwork; Gemini/GPTselfcomputed,Claudeproviderreports","family":"BioMysteryBench","locator":"Model card page5 Results table","metric":"percent","notes":"","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":56.5,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"biomysterybench-human-difficult","evidenceKind":"lab_self_report","harnessId":"biomysterybench-human-difficult:google-gemini-3-8-card:TGludXh0ZXJtaW5hbCxiaW9pbmZvdG9vbHMsUHl0aG9uLFI7IGFsbG93bGlzdGVkbmV0d29yazsgR2VtaW5pL0dQVHNlbGZjb21wdXRlZCxDbGF1ZGVwcm92aWRlcnJlcG9ydHM","id":"launch-google-gemini-3-8-card-2095","ingestRunId":"launch-google-gemini-3-8-card","modelId":"gemini-3.7-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Linuxterminal,bioinfotools,Python,R; allowlistednetwork; Gemini/GPTselfcomputed,Claudeproviderreports","family":"BioMysteryBench","locator":"Model card page5 Results table","metric":"percent","notes":"","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":43.5,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"biomysterybench-human-difficult","evidenceKind":"lab_self_report","harnessId":"biomysterybench-human-difficult:google-gemini-3-8-card:TGludXh0ZXJtaW5hbCxiaW9pbmZvdG9vbHMsUHl0aG9uLFI7IGFsbG93bGlzdGVkbmV0d29yazsgR2VtaW5pL0dQVHNlbGZjb21wdXRlZCxDbGF1ZGVwcm92aWRlcnJlcG9ydHM","id":"launch-google-gemini-3-8-card-2096","ingestRunId":"launch-google-gemini-3-8-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Linuxterminal,bioinfotools,Python,R; allowlistednetwork; Gemini/GPTselfcomputed,Claudeproviderreports","family":"BioMysteryBench","locator":"Model card page5 Results table","metric":"percent","notes":"","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":49.4,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"biomysterybench-human-difficult","evidenceKind":"lab_self_report","harnessId":"biomysterybench-human-difficult:google-gemini-3-8-card:TGludXh0ZXJtaW5hbCxiaW9pbmZvdG9vbHMsUHl0aG9uLFI7IGFsbG93bGlzdGVkbmV0d29yazsgR2VtaW5pL0dQVHNlbGZjb21wdXRlZCxDbGF1ZGVwcm92aWRlcnJlcG9ydHM","id":"launch-google-gemini-3-8-card-2097","ingestRunId":"launch-google-gemini-3-8-card","modelId":"claude-sonnet-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Linuxterminal,bioinfotools,Python,R; allowlistednetwork; Gemini/GPTselfcomputed,Claudeproviderreports","family":"BioMysteryBench","locator":"Model card page5 Results table","metric":"percent","notes":"","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":34.1,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"biomysterybench-human-difficult","evidenceKind":"lab_self_report","harnessId":"biomysterybench-human-difficult:google-gemini-3-8-card:TGludXh0ZXJtaW5hbCxiaW9pbmZvdG9vbHMsUHl0aG9uLFI7IGFsbG93bGlzdGVkbmV0d29yazsgR2VtaW5pL0dQVHNlbGZjb21wdXRlZCxDbGF1ZGVwcm92aWRlcnJlcG9ydHM","id":"launch-google-gemini-3-8-card-2098","ingestRunId":"launch-google-gemini-3-8-card","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Linuxterminal,bioinfotools,Python,R; allowlistednetwork; Gemini/GPTselfcomputed,Claudeproviderreports","family":"BioMysteryBench","locator":"Model card page5 Results table","metric":"percent","notes":"","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":44.7,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"biomysterybench-human-difficult","evidenceKind":"lab_self_report","harnessId":"biomysterybench-human-difficult:google-gemini-3-8-card:TGludXh0ZXJtaW5hbCxiaW9pbmZvdG9vbHMsUHl0aG9uLFI7IGFsbG93bGlzdGVkbmV0d29yazsgR2VtaW5pL0dQVHNlbGZjb21wdXRlZCxDbGF1ZGVwcm92aWRlcnJlcG9ydHM","id":"launch-google-gemini-3-8-card-2099","ingestRunId":"launch-google-gemini-3-8-card","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Linuxterminal,bioinfotools,Python,R; allowlistednetwork; Gemini/GPTselfcomputed,Claudeproviderreports","family":"BioMysteryBench","locator":"Model card page5 Results table","metric":"percent","notes":"","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":49.4,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"labbench-2","evidenceKind":"lab_self_report","harnessId":"labbench-2:google-gemini-3-8-card:U2VsZmNvbXB1dGVkOyBMaW51eHRlcm1pbmFsLGJpb2luZm90b29scyxQeXRob24sUixuZXR3b3JrOyBtYWNyb2F2ZXJhZ2UxMXN1YnRhc2tz","id":"launch-google-gemini-3-8-card-2100","ingestRunId":"launch-google-gemini-3-8-card","modelId":"gemini-3.8-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Selfcomputed; Linuxterminal,bioinfotools,Python,R,network; macroaverage11subtasks","family":"LABBench","locator":"Model card page5 Results table","metric":"percent","notes":"","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":86.2,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"labbench-2","evidenceKind":"lab_self_report","harnessId":"labbench-2:google-gemini-3-8-card:U2VsZmNvbXB1dGVkOyBMaW51eHRlcm1pbmFsLGJpb2luZm90b29scyxQeXRob24sUixuZXR3b3JrOyBtYWNyb2F2ZXJhZ2UxMXN1YnRhc2tz","id":"launch-google-gemini-3-8-card-2101","ingestRunId":"launch-google-gemini-3-8-card","modelId":"gemini-3.7-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Selfcomputed; Linuxterminal,bioinfotools,Python,R,network; macroaverage11subtasks","family":"LABBench","locator":"Model card page5 Results table","metric":"percent","notes":"","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":82.1,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"labbench-2","evidenceKind":"lab_self_report","harnessId":"labbench-2:google-gemini-3-8-card:U2VsZmNvbXB1dGVkOyBMaW51eHRlcm1pbmFsLGJpb2luZm90b29scyxQeXRob24sUixuZXR3b3JrOyBtYWNyb2F2ZXJhZ2UxMXN1YnRhc2tz","id":"launch-google-gemini-3-8-card-2102","ingestRunId":"launch-google-gemini-3-8-card","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Selfcomputed; Linuxterminal,bioinfotools,Python,R,network; macroaverage11subtasks","family":"LABBench","locator":"Model card page5 Results table","metric":"percent","notes":"","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":84.2,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"labbench-2","evidenceKind":"lab_self_report","harnessId":"labbench-2:google-gemini-3-8-card:U2VsZmNvbXB1dGVkOyBMaW51eHRlcm1pbmFsLGJpb2luZm90b29scyxQeXRob24sUixuZXR3b3JrOyBtYWNyb2F2ZXJhZ2UxMXN1YnRhc2tz","id":"launch-google-gemini-3-8-card-2103","ingestRunId":"launch-google-gemini-3-8-card","modelId":"claude-sonnet-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Selfcomputed; Linuxterminal,bioinfotools,Python,R,network; macroaverage11subtasks","family":"LABBench","locator":"Model card page5 Results table","metric":"percent","notes":"","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":80.1,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"labbench-2","evidenceKind":"lab_self_report","harnessId":"labbench-2:google-gemini-3-8-card:U2VsZmNvbXB1dGVkOyBMaW51eHRlcm1pbmFsLGJpb2luZm90b29scyxQeXRob24sUixuZXR3b3JrOyBtYWNyb2F2ZXJhZ2UxMXN1YnRhc2tz","id":"launch-google-gemini-3-8-card-2104","ingestRunId":"launch-google-gemini-3-8-card","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Selfcomputed; Linuxterminal,bioinfotools,Python,R,network; macroaverage11subtasks","family":"LABBench","locator":"Model card page5 Results table","metric":"percent","notes":"","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":82.1,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"labbench-2","evidenceKind":"lab_self_report","harnessId":"labbench-2:google-gemini-3-8-card:U2VsZmNvbXB1dGVkOyBMaW51eHRlcm1pbmFsLGJpb2luZm90b29scyxQeXRob24sUixuZXR3b3JrOyBtYWNyb2F2ZXJhZ2UxMXN1YnRhc2tz","id":"launch-google-gemini-3-8-card-2105","ingestRunId":"launch-google-gemini-3-8-card","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Selfcomputed; Linuxterminal,bioinfotools,Python,R,network; macroaverage11subtasks","family":"LABBench","locator":"Model card page5 Results table","metric":"percent","notes":"","sourceId":"google-gemini-3-8-card","sourceTitle":"Gemini3.8Flash Model Card"},"score":81.2,"scoreUnit":"percent","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf"},{"benchmarkId":"gpqa-diamond-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gpqa-diamond-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-702","ingestRunId":"launch-kimi","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"hard_reasoning","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"GPQA Diamond","locator":"Performance table: GPQA Diamond","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":93.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"gpqa-diamond-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gpqa-diamond-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-703","ingestRunId":"launch-kimi","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"hard_reasoning","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"GPQA Diamond","locator":"Performance table: GPQA Diamond","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":92.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"gpqa-diamond-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gpqa-diamond-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-704","ingestRunId":"launch-kimi","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"hard_reasoning","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"GPQA Diamond","locator":"Performance table: GPQA Diamond","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":94.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"gpqa-diamond-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gpqa-diamond-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-705","ingestRunId":"launch-kimi","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"hard_reasoning","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"GPQA Diamond","locator":"Performance table: GPQA Diamond","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":91,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"gpqa-diamond-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gpqa-diamond-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-706","ingestRunId":"launch-kimi","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"hard_reasoning","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"GPQA Diamond","locator":"Performance table: GPQA Diamond","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":93.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"gpqa-diamond-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gpqa-diamond-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-707","ingestRunId":"launch-kimi","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"hard_reasoning","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"GPQA Diamond","locator":"Performance table: GPQA Diamond","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":91.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"critpt-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"critpt-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-708","ingestRunId":"launch-kimi","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"hard_reasoning","comparabilityReason":"Kimi K3 Evaluation Details cites Artificial Analysis; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"CritPt","locator":"Performance table: CritPt","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":23.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"critpt-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"critpt-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-709","ingestRunId":"launch-kimi","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"hard_reasoning","comparabilityReason":"Kimi K3 Evaluation Details cites Artificial Analysis; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"CritPt","locator":"Performance table: CritPt","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":28.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"critpt-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"critpt-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-710","ingestRunId":"launch-kimi","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"hard_reasoning","comparabilityReason":"Kimi K3 Evaluation Details cites Artificial Analysis; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"CritPt","locator":"Performance table: CritPt","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":32.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"critpt-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"critpt-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-711","ingestRunId":"launch-kimi","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"hard_reasoning","comparabilityReason":"Kimi K3 Evaluation Details cites Artificial Analysis; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"CritPt","locator":"Performance table: CritPt","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":20.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"critpt-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"critpt-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-712","ingestRunId":"launch-kimi","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"hard_reasoning","comparabilityReason":"Kimi K3 Evaluation Details cites Artificial Analysis; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"CritPt","locator":"Performance table: CritPt","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":27.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"critpt-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"critpt-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-713","ingestRunId":"launch-kimi","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"hard_reasoning","comparabilityReason":"Kimi K3 Evaluation Details cites Artificial Analysis; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"CritPt","locator":"Performance table: CritPt","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":20.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"aa-lcr-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"aa-lcr-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-714","ingestRunId":"launch-kimi","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"long_context","comparabilityReason":"Kimi K3 Evaluation Details cites Artificial Analysis; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"AA-LCR","locator":"Performance table: AA-LCR","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":74.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"aa-lcr-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"aa-lcr-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-715","ingestRunId":"launch-kimi","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"long_context","comparabilityReason":"Kimi K3 Evaluation Details cites Artificial Analysis; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"AA-LCR","locator":"Performance table: AA-LCR","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":70,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"aa-lcr-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"aa-lcr-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-716","ingestRunId":"launch-kimi","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"long_context","comparabilityReason":"Kimi K3 Evaluation Details cites Artificial Analysis; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"AA-LCR","locator":"Performance table: AA-LCR","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":73.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"aa-lcr-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"aa-lcr-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-717","ingestRunId":"launch-kimi","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"long_context","comparabilityReason":"Kimi K3 Evaluation Details cites Artificial Analysis; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"AA-LCR","locator":"Performance table: AA-LCR","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":67.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"aa-lcr-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"aa-lcr-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-718","ingestRunId":"launch-kimi","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"long_context","comparabilityReason":"Kimi K3 Evaluation Details cites Artificial Analysis; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"AA-LCR","locator":"Performance table: AA-LCR","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":74.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"aa-lcr-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"aa-lcr-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-719","ingestRunId":"launch-kimi","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"long_context","comparabilityReason":"Kimi K3 Evaluation Details cites Artificial Analysis; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"AA-LCR","locator":"Performance table: AA-LCR","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":71.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"hle-full-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hle-full-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aG91dCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","id":"launch-kimi-720","ingestRunId":"launch-kimi","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"hard_reasoning","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Humanity's Last Exam","locator":"Performance table: HLE-Full","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":43.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"hle-full-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hle-full-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","id":"launch-kimi-721","ingestRunId":"launch-kimi","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"hard_reasoning","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Humanity's Last Exam","locator":"Performance table: HLE-Full","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":56,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"hle-full-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hle-full-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aG91dCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","id":"launch-kimi-722","ingestRunId":"launch-kimi","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"hard_reasoning","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Humanity's Last Exam","locator":"Performance table: HLE-Full","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":53.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"hle-full-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hle-full-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","id":"launch-kimi-723","ingestRunId":"launch-kimi","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"hard_reasoning","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Humanity's Last Exam","locator":"Performance table: HLE-Full","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":63,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"hle-full-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hle-full-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aG91dCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","id":"launch-kimi-724","ingestRunId":"launch-kimi","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"hard_reasoning","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Humanity's Last Exam","locator":"Performance table: HLE-Full","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":44.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"hle-full-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hle-full-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","id":"launch-kimi-725","ingestRunId":"launch-kimi","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"hard_reasoning","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Humanity's Last Exam","locator":"Performance table: HLE-Full","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":58,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"hle-full-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hle-full-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aG91dCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","id":"launch-kimi-726","ingestRunId":"launch-kimi","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"hard_reasoning","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Humanity's Last Exam","locator":"Performance table: HLE-Full","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":49.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"hle-full-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hle-full-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","id":"launch-kimi-727","ingestRunId":"launch-kimi","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"hard_reasoning","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Humanity's Last Exam","locator":"Performance table: HLE-Full","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":57.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"hle-full-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hle-full-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aG91dCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","id":"launch-kimi-728","ingestRunId":"launch-kimi","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"hard_reasoning","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Humanity's Last Exam","locator":"Performance table: HLE-Full","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":41.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"hle-full-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hle-full-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","id":"launch-kimi-729","ingestRunId":"launch-kimi","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"hard_reasoning","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Humanity's Last Exam","locator":"Performance table: HLE-Full","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":52.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"deepswe-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"deepswe-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-730","ingestRunId":"launch-kimi","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"DeepSWE","locator":"Performance table: DeepSWE","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":67.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"deepswe-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"deepswe-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-731","ingestRunId":"launch-kimi","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"DeepSWE","locator":"Performance table: DeepSWE","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":70,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"deepswe-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"deepswe-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-732","ingestRunId":"launch-kimi","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"DeepSWE","locator":"Performance table: DeepSWE","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":73,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"deepswe-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"deepswe-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-733","ingestRunId":"launch-kimi","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"DeepSWE","locator":"Performance table: DeepSWE","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":59,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"deepswe-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"deepswe-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-734","ingestRunId":"launch-kimi","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"DeepSWE","locator":"Performance table: DeepSWE","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":67,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"deepswe-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"deepswe-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-735","ingestRunId":"launch-kimi","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"DeepSWE","locator":"Performance table: DeepSWE","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":46.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"programbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"programbench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-736","ingestRunId":"launch-kimi","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"ProgramBench","locator":"Performance table: ProgramBench","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":77.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"programbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"programbench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-737","ingestRunId":"launch-kimi","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","comparabilityReason":"Kimi K3 Evaluation Details cites GLM release blog or Vals AI, per model; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"ProgramBench","locator":"Performance table: ProgramBench","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":76.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"programbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"programbench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-738","ingestRunId":"launch-kimi","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","comparabilityReason":"Kimi K3 Evaluation Details cites GLM release blog or Vals AI, per model; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"ProgramBench","locator":"Performance table: ProgramBench","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":77.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"programbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"programbench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-739","ingestRunId":"launch-kimi","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","comparabilityReason":"Kimi K3 Evaluation Details cites GLM release blog or Vals AI, per model; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"ProgramBench","locator":"Performance table: ProgramBench","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":71.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"programbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"programbench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-740","ingestRunId":"launch-kimi","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","comparabilityReason":"Kimi K3 Evaluation Details cites GLM release blog or Vals AI, per model; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"ProgramBench","locator":"Performance table: ProgramBench","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":70.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"programbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"programbench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-741","ingestRunId":"launch-kimi","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","comparabilityReason":"Kimi K3 Evaluation Details cites GLM release blog or Vals AI, per model; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"ProgramBench","locator":"Performance table: ProgramBench","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":63.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"terminal-bench-2-1-pct","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1-pct:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-742","ingestRunId":"launch-kimi","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Terminal-Bench","locator":"Performance table: Terminal-Bench 2.1","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":88.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"terminal-bench-2-1-pct","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1-pct:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-743","ingestRunId":"launch-kimi","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","comparabilityReason":"Kimi K3 Evaluation Details cites GLM release blog, Artificial Analysis or OpenAI, per model; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Terminal-Bench","locator":"Performance table: Terminal-Bench 2.1","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":88,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"terminal-bench-2-1-pct","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1-pct:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-744","ingestRunId":"launch-kimi","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","comparabilityReason":"Kimi K3 Evaluation Details cites GLM release blog, Artificial Analysis or OpenAI, per model; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Terminal-Bench","locator":"Performance table: Terminal-Bench 2.1","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":88.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"terminal-bench-2-1-pct","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1-pct:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-745","ingestRunId":"launch-kimi","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","comparabilityReason":"Kimi K3 Evaluation Details cites GLM release blog, Artificial Analysis or OpenAI, per model; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Terminal-Bench","locator":"Performance table: Terminal-Bench 2.1","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":84.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"terminal-bench-2-1-pct","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1-pct:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-746","ingestRunId":"launch-kimi","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","comparabilityReason":"Kimi K3 Evaluation Details cites GLM release blog, Artificial Analysis or OpenAI, per model; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Terminal-Bench","locator":"Performance table: Terminal-Bench 2.1","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":83.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"terminal-bench-2-1-pct","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1-pct:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-747","ingestRunId":"launch-kimi","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","comparabilityReason":"Kimi K3 Evaluation Details cites GLM release blog, Artificial Analysis or OpenAI, per model; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Terminal-Bench","locator":"Performance table: Terminal-Bench 2.1","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":82.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"frontierswe-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"frontierswe-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-748","ingestRunId":"launch-kimi","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"FrontierSWE","locator":"Performance table: FrontierSWE","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":81.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"frontierswe-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"frontierswe-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-749","ingestRunId":"launch-kimi","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","comparabilityReason":"Kimi K3 Evaluation Details cites FrontierSWE leaderboard; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"FrontierSWE","locator":"Performance table: FrontierSWE","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":86.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"frontierswe-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"frontierswe-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-750","ingestRunId":"launch-kimi","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"FrontierSWE","locator":"Performance table: FrontierSWE","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":71.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"frontierswe-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"frontierswe-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-751","ingestRunId":"launch-kimi","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","comparabilityReason":"Kimi K3 Evaluation Details cites FrontierSWE leaderboard; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"FrontierSWE","locator":"Performance table: FrontierSWE","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":66.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"frontierswe-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"frontierswe-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-752","ingestRunId":"launch-kimi","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","comparabilityReason":"Kimi K3 Evaluation Details cites FrontierSWE leaderboard; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"FrontierSWE","locator":"Performance table: FrontierSWE","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":64.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"frontierswe-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"frontierswe-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-753","ingestRunId":"launch-kimi","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","comparabilityReason":"Kimi K3 Evaluation Details cites FrontierSWE leaderboard; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"FrontierSWE","locator":"Performance table: FrontierSWE","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":67.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"swe-marathon-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"swe-marathon-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-754","ingestRunId":"launch-kimi","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"SWE-Marathon","locator":"Performance table: SWE-Marathon","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":42,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"swe-marathon-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"swe-marathon-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-755","ingestRunId":"launch-kimi","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"SWE-Marathon","locator":"Performance table: SWE-Marathon","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":35,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"swe-marathon-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"swe-marathon-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-756","ingestRunId":"launch-kimi","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"SWE-Marathon","locator":"Performance table: SWE-Marathon","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":39,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"swe-marathon-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"swe-marathon-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-757","ingestRunId":"launch-kimi","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"SWE-Marathon","locator":"Performance table: SWE-Marathon","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":40,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"swe-marathon-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"swe-marathon-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-758","ingestRunId":"launch-kimi","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"SWE-Marathon","locator":"Performance table: SWE-Marathon","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":14,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"swe-marathon-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"swe-marathon-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-759","ingestRunId":"launch-kimi","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","comparabilityReason":"Kimi K3 Evaluation Details cites GLM-5.2 release blog; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"SWE-Marathon","locator":"Performance table: SWE-Marathon","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":13,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"posttrainbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"posttrainbench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-760","ingestRunId":"launch-kimi","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"PostTrainBench","locator":"Performance table: PostTrainBench","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":36.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"posttrainbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"posttrainbench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-761","ingestRunId":"launch-kimi","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"PostTrainBench","locator":"Performance table: PostTrainBench","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":41.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"posttrainbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"posttrainbench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-762","ingestRunId":"launch-kimi","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"PostTrainBench","locator":"Performance table: PostTrainBench","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":34.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"posttrainbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"posttrainbench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-763","ingestRunId":"launch-kimi","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","comparabilityReason":"Kimi K3 Evaluation Details cites official PostTrainBench results; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"PostTrainBench","locator":"Performance table: PostTrainBench","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":34.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"posttrainbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"posttrainbench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-764","ingestRunId":"launch-kimi","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","comparabilityReason":"Kimi K3 Evaluation Details cites official PostTrainBench results; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"PostTrainBench","locator":"Performance table: PostTrainBench","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":28.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"posttrainbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"posttrainbench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-765","ingestRunId":"launch-kimi","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","comparabilityReason":"Kimi K3 Evaluation Details cites official PostTrainBench results; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"PostTrainBench","locator":"Performance table: PostTrainBench","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":34.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"mls-bench-lite-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mls-bench-lite-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-766","ingestRunId":"launch-kimi","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"MLS-Bench-Lite","locator":"Performance table: MLS-Bench-Lite","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":48.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"mls-bench-lite-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mls-bench-lite-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-767","ingestRunId":"launch-kimi","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"MLS-Bench-Lite","locator":"Performance table: MLS-Bench-Lite","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":49.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"mls-bench-lite-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mls-bench-lite-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-768","ingestRunId":"launch-kimi","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"MLS-Bench-Lite","locator":"Performance table: MLS-Bench-Lite","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":46.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"mls-bench-lite-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mls-bench-lite-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-769","ingestRunId":"launch-kimi","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"MLS-Bench-Lite","locator":"Performance table: MLS-Bench-Lite","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":42.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"mls-bench-lite-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mls-bench-lite-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-770","ingestRunId":"launch-kimi","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"MLS-Bench-Lite","locator":"Performance table: MLS-Bench-Lite","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":35.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"mls-bench-lite-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mls-bench-lite-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-771","ingestRunId":"launch-kimi","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"MLS-Bench-Lite","locator":"Performance table: MLS-Bench-Lite","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":40.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"scicode-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"scicode-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-772","ingestRunId":"launch-kimi","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","comparabilityReason":"Kimi K3 Evaluation Details cites Artificial Analysis; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"SciCode","locator":"Performance table: SciCode","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":58.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"scicode-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"scicode-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-773","ingestRunId":"launch-kimi","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","comparabilityReason":"Kimi K3 Evaluation Details cites Artificial Analysis; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"SciCode","locator":"Performance table: SciCode","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":60.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"scicode-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"scicode-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-774","ingestRunId":"launch-kimi","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","comparabilityReason":"Kimi K3 Evaluation Details cites Artificial Analysis; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"SciCode","locator":"Performance table: SciCode","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":56.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"scicode-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"scicode-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-775","ingestRunId":"launch-kimi","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","comparabilityReason":"Kimi K3 Evaluation Details cites Artificial Analysis; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"SciCode","locator":"Performance table: SciCode","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":53.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"scicode-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"scicode-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-776","ingestRunId":"launch-kimi","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","comparabilityReason":"Kimi K3 Evaluation Details cites Artificial Analysis; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"SciCode","locator":"Performance table: SciCode","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":56.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"scicode-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"scicode-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-777","ingestRunId":"launch-kimi","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","comparabilityReason":"Kimi K3 Evaluation Details cites Artificial Analysis; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"SciCode","locator":"Performance table: SciCode","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":50.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"kimi-code-bench-2-0","evidenceKind":"lab_self_report","harnessId":"kimi-code-bench-2-0:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-778","ingestRunId":"launch-kimi","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Kimi Code Bench 2.0","locator":"Performance table: Kimi Code Bench 2.0","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":72.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"kimi-code-bench-2-0","evidenceKind":"lab_self_report","harnessId":"kimi-code-bench-2-0:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-779","ingestRunId":"launch-kimi","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Kimi Code Bench 2.0","locator":"Performance table: Kimi Code Bench 2.0","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":76.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"kimi-code-bench-2-0","evidenceKind":"lab_self_report","harnessId":"kimi-code-bench-2-0:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-780","ingestRunId":"launch-kimi","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Kimi Code Bench 2.0","locator":"Performance table: Kimi Code Bench 2.0","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":64.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"kimi-code-bench-2-0","evidenceKind":"lab_self_report","harnessId":"kimi-code-bench-2-0:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-781","ingestRunId":"launch-kimi","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Kimi Code Bench 2.0","locator":"Performance table: Kimi Code Bench 2.0","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":71.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"kimi-code-bench-2-0","evidenceKind":"lab_self_report","harnessId":"kimi-code-bench-2-0:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-782","ingestRunId":"launch-kimi","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Kimi Code Bench 2.0","locator":"Performance table: Kimi Code Bench 2.0","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":69,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"kimi-code-bench-2-0","evidenceKind":"lab_self_report","harnessId":"kimi-code-bench-2-0:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-783","ingestRunId":"launch-kimi","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Kimi Code Bench 2.0","locator":"Performance table: Kimi Code Bench 2.0","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":64.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"browsecomp-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"browsecomp-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-784","ingestRunId":"launch-kimi","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"BrowseComp","locator":"Performance table: BrowseComp","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":91.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"browsecomp-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"browsecomp-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-785","ingestRunId":"launch-kimi","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites Anthropic or OpenAI release reports; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"BrowseComp","locator":"Performance table: BrowseComp","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":88,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"browsecomp-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"browsecomp-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-786","ingestRunId":"launch-kimi","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites Anthropic or OpenAI release reports; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"BrowseComp","locator":"Performance table: BrowseComp","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":90.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"browsecomp-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"browsecomp-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-787","ingestRunId":"launch-kimi","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites Anthropic or OpenAI release reports; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"BrowseComp","locator":"Performance table: BrowseComp","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":84.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"browsecomp-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"browsecomp-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-788","ingestRunId":"launch-kimi","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites Anthropic or OpenAI release reports; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"BrowseComp","locator":"Performance table: BrowseComp","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":84.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"deepsearchqa-f1-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"deepsearchqa-f1-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-789","ingestRunId":"launch-kimi","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"DeepSearchQA (F1)","locator":"Performance table: DeepSearchQA (F1)","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":95,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"deepsearchqa-f1-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"deepsearchqa-f1-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-790","ingestRunId":"launch-kimi","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"DeepSearchQA (F1)","locator":"Performance table: DeepSearchQA (F1)","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":94.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"deepsearchqa-f1-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"deepsearchqa-f1-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-791","ingestRunId":"launch-kimi","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"DeepSearchQA (F1)","locator":"Performance table: DeepSearchQA (F1)","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":93.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"researchrubrics-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"researchrubrics-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-792","ingestRunId":"launch-kimi","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"ResearchRubrics","locator":"Performance table: ResearchRubrics","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":76.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"researchrubrics-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"researchrubrics-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-793","ingestRunId":"launch-kimi","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"ResearchRubrics","locator":"Performance table: ResearchRubrics","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":73.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"researchrubrics-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"researchrubrics-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-794","ingestRunId":"launch-kimi","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"ResearchRubrics","locator":"Performance table: ResearchRubrics","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":73.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"researchrubrics-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"researchrubrics-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-795","ingestRunId":"launch-kimi","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"ResearchRubrics","locator":"Performance table: ResearchRubrics","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":64,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"researchrubrics-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"researchrubrics-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-796","ingestRunId":"launch-kimi","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"ResearchRubrics","locator":"Performance table: ResearchRubrics","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":71.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"gdpval-aa-v2-elo-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gdpval-aa-v2-elo-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-797","ingestRunId":"launch-kimi","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites Artificial Analysis; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"GDPval-AA v2 (Elo)","locator":"Performance table: GDPval-AA v2 (Elo)","metric":"Elo","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":1686,"scoreUnit":"elo","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"gdpval-aa-v2-elo-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gdpval-aa-v2-elo-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-798","ingestRunId":"launch-kimi","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites Artificial Analysis; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"GDPval-AA v2 (Elo)","locator":"Performance table: GDPval-AA v2 (Elo)","metric":"Elo","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":1747,"scoreUnit":"elo","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"gdpval-aa-v2-elo-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gdpval-aa-v2-elo-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-799","ingestRunId":"launch-kimi","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites Artificial Analysis; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"GDPval-AA v2 (Elo)","locator":"Performance table: GDPval-AA v2 (Elo)","metric":"Elo","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":1736,"scoreUnit":"elo","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"gdpval-aa-v2-elo-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gdpval-aa-v2-elo-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-800","ingestRunId":"launch-kimi","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites Artificial Analysis; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"GDPval-AA v2 (Elo)","locator":"Performance table: GDPval-AA v2 (Elo)","metric":"Elo","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":1593,"scoreUnit":"elo","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"gdpval-aa-v2-elo-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gdpval-aa-v2-elo-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-801","ingestRunId":"launch-kimi","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites Artificial Analysis; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"GDPval-AA v2 (Elo)","locator":"Performance table: GDPval-AA v2 (Elo)","metric":"Elo","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":1491,"scoreUnit":"elo","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"gdpval-aa-v2-elo-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gdpval-aa-v2-elo-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-802","ingestRunId":"launch-kimi","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites Artificial Analysis; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"GDPval-AA v2 (Elo)","locator":"Performance table: GDPval-AA v2 (Elo)","metric":"Elo","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":1510,"scoreUnit":"elo","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"toolathlon-verified-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"toolathlon-verified-source-release-snapshot-version-not-specified-pct:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-803","ingestRunId":"launch-kimi","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Toolathlon-Verified","locator":"Performance table: Toolathlon-Verified","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":76.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"toolathlon-verified-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"toolathlon-verified-source-release-snapshot-version-not-specified-pct:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-804","ingestRunId":"launch-kimi","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Toolathlon-Verified","locator":"Performance table: Toolathlon-Verified","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":77.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"toolathlon-verified-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"toolathlon-verified-source-release-snapshot-version-not-specified-pct:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-805","ingestRunId":"launch-kimi","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Toolathlon-Verified","locator":"Performance table: Toolathlon-Verified","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":74.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"toolathlon-verified-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"toolathlon-verified-source-release-snapshot-version-not-specified-pct:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-806","ingestRunId":"launch-kimi","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Toolathlon-Verified","locator":"Performance table: Toolathlon-Verified","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":76.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"toolathlon-verified-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"toolathlon-verified-source-release-snapshot-version-not-specified-pct:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-807","ingestRunId":"launch-kimi","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Toolathlon-Verified","locator":"Performance table: Toolathlon-Verified","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":73.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"toolathlon-verified-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"toolathlon-verified-source-release-snapshot-version-not-specified-pct:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-808","ingestRunId":"launch-kimi","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Toolathlon-Verified","locator":"Performance table: Toolathlon-Verified","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":59.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"mcpmark-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mcpmark-verified-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-809","ingestRunId":"launch-kimi","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"MCPMark-Verified","locator":"Performance table: MCPMark-Verified","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":94.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"mcpmark-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mcpmark-verified-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-810","ingestRunId":"launch-kimi","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"MCPMark-Verified","locator":"Performance table: MCPMark-Verified","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":87.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"mcpmark-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mcpmark-verified-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-811","ingestRunId":"launch-kimi","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"MCPMark-Verified","locator":"Performance table: MCPMark-Verified","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":92.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"mcpmark-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mcpmark-verified-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-812","ingestRunId":"launch-kimi","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"MCPMark-Verified","locator":"Performance table: MCPMark-Verified","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":76.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"mcpmark-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mcpmark-verified-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-813","ingestRunId":"launch-kimi","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"MCPMark-Verified","locator":"Performance table: MCPMark-Verified","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":92.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"mcp-atlas-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mcp-atlas-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-814","ingestRunId":"launch-kimi","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"MCP-Atlas","locator":"Performance table: MCP-Atlas","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":84.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"mcp-atlas-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mcp-atlas-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-815","ingestRunId":"launch-kimi","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"MCP-Atlas","locator":"Performance table: MCP-Atlas","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":84.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"mcp-atlas-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mcp-atlas-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-816","ingestRunId":"launch-kimi","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"MCP-Atlas","locator":"Performance table: MCP-Atlas","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":83.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"mcp-atlas-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mcp-atlas-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-817","ingestRunId":"launch-kimi","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"MCP-Atlas","locator":"Performance table: MCP-Atlas","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":83.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"mcp-atlas-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mcp-atlas-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-818","ingestRunId":"launch-kimi","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"MCP-Atlas","locator":"Performance table: MCP-Atlas","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":82.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"mcp-atlas-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mcp-atlas-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-819","ingestRunId":"launch-kimi","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"MCP-Atlas","locator":"Performance table: MCP-Atlas","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":82.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"automationbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"automationbench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-820","ingestRunId":"launch-kimi","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"AutomationBench","locator":"Performance table: AutomationBench","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":30.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"automationbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"automationbench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-821","ingestRunId":"launch-kimi","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"AutomationBench","locator":"Performance table: AutomationBench","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":29.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"automationbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"automationbench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-822","ingestRunId":"launch-kimi","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"AutomationBench","locator":"Performance table: AutomationBench","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":29.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"automationbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"automationbench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-823","ingestRunId":"launch-kimi","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"AutomationBench","locator":"Performance table: AutomationBench","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":27.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"automationbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"automationbench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-824","ingestRunId":"launch-kimi","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"AutomationBench","locator":"Performance table: AutomationBench","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":22.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"automationbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"automationbench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-825","ingestRunId":"launch-kimi","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"AutomationBench","locator":"Performance table: AutomationBench","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":12.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"jobbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"jobbench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-826","ingestRunId":"launch-kimi","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"JobBench","locator":"Performance table: JobBench","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":54.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"jobbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"jobbench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-827","ingestRunId":"launch-kimi","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"JobBench","locator":"Performance table: JobBench","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":57.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"jobbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"jobbench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-828","ingestRunId":"launch-kimi","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"JobBench","locator":"Performance table: JobBench","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":45.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"jobbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"jobbench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-829","ingestRunId":"launch-kimi","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"JobBench","locator":"Performance table: JobBench","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":48.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"jobbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"jobbench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-830","ingestRunId":"launch-kimi","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"JobBench","locator":"Performance table: JobBench","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":38.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"jobbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"jobbench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-831","ingestRunId":"launch-kimi","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"JobBench","locator":"Performance table: JobBench","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":43.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"aa-briefcase-elo-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"aa-briefcase-elo-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-832","ingestRunId":"launch-kimi","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites Artificial Analysis; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"AA-Briefcase (Elo)","locator":"Performance table: AA-Briefcase (Elo)","metric":"Elo","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":1548,"scoreUnit":"elo","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"aa-briefcase-elo-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"aa-briefcase-elo-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-833","ingestRunId":"launch-kimi","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites Artificial Analysis; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"AA-Briefcase (Elo)","locator":"Performance table: AA-Briefcase (Elo)","metric":"Elo","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":1583,"scoreUnit":"elo","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"aa-briefcase-elo-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"aa-briefcase-elo-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-834","ingestRunId":"launch-kimi","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites Artificial Analysis; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"AA-Briefcase (Elo)","locator":"Performance table: AA-Briefcase (Elo)","metric":"Elo","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":1495,"scoreUnit":"elo","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"aa-briefcase-elo-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"aa-briefcase-elo-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-835","ingestRunId":"launch-kimi","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites Artificial Analysis; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"AA-Briefcase (Elo)","locator":"Performance table: AA-Briefcase (Elo)","metric":"Elo","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":1354,"scoreUnit":"elo","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"aa-briefcase-elo-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"aa-briefcase-elo-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-836","ingestRunId":"launch-kimi","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites Artificial Analysis; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"AA-Briefcase (Elo)","locator":"Performance table: AA-Briefcase (Elo)","metric":"Elo","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":1158,"scoreUnit":"elo","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"aa-briefcase-elo-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"aa-briefcase-elo-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-837","ingestRunId":"launch-kimi","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites Artificial Analysis; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"AA-Briefcase (Elo)","locator":"Performance table: AA-Briefcase (Elo)","metric":"Elo","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":1260,"scoreUnit":"elo","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"agents-last-exam-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"agents-last-exam-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-838","ingestRunId":"launch-kimi","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites official Agents Last Exam leaderboard; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Agents' Last Exam","locator":"Performance table: Agents' Last Exam","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":28.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"agents-last-exam-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"agents-last-exam-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-839","ingestRunId":"launch-kimi","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites official Agents Last Exam leaderboard; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Agents' Last Exam","locator":"Performance table: Agents' Last Exam","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":25.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"agents-last-exam-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"agents-last-exam-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-840","ingestRunId":"launch-kimi","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites official Agents Last Exam leaderboard; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Agents' Last Exam","locator":"Performance table: Agents' Last Exam","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":29.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"agents-last-exam-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"agents-last-exam-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-841","ingestRunId":"launch-kimi","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites official Agents Last Exam leaderboard; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Agents' Last Exam","locator":"Performance table: Agents' Last Exam","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":27,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"agents-last-exam-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"agents-last-exam-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-842","ingestRunId":"launch-kimi","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites official Agents Last Exam leaderboard; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Agents' Last Exam","locator":"Performance table: Agents' Last Exam","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":26.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"agents-last-exam-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"agents-last-exam-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-843","ingestRunId":"launch-kimi","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites official Agents Last Exam leaderboard; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Agents' Last Exam","locator":"Performance table: Agents' Last Exam","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":20.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"apex-agents-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"apex-agents-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-844","ingestRunId":"launch-kimi","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites Artificial Analysis / APEX-Agents leaderboard; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"APEX-Agents","locator":"Performance table: APEX-Agents","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":41,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"apex-agents-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"apex-agents-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-845","ingestRunId":"launch-kimi","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites Artificial Analysis / APEX-Agents leaderboard; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"APEX-Agents","locator":"Performance table: APEX-Agents","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":43.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"apex-agents-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"apex-agents-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-846","ingestRunId":"launch-kimi","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites Artificial Analysis / APEX-Agents leaderboard; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"APEX-Agents","locator":"Performance table: APEX-Agents","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":39.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"apex-agents-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"apex-agents-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-847","ingestRunId":"launch-kimi","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites Artificial Analysis / APEX-Agents leaderboard; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"APEX-Agents","locator":"Performance table: APEX-Agents","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":39.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"apex-agents-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"apex-agents-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-848","ingestRunId":"launch-kimi","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites Artificial Analysis / APEX-Agents leaderboard; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"APEX-Agents","locator":"Performance table: APEX-Agents","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":38.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"apex-agents-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"apex-agents-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-849","ingestRunId":"launch-kimi","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites Artificial Analysis / APEX-Agents leaderboard; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"APEX-Agents","locator":"Performance table: APEX-Agents","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":35.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"officeqa-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"officeqa-pro-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-850","ingestRunId":"launch-kimi","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"OfficeQA Pro","locator":"Performance table: OfficeQA Pro","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":63.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"officeqa-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"officeqa-pro-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-851","ingestRunId":"launch-kimi","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"OfficeQA Pro","locator":"Performance table: OfficeQA Pro","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":69.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"officeqa-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"officeqa-pro-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-852","ingestRunId":"launch-kimi","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"OfficeQA Pro","locator":"Performance table: OfficeQA Pro","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":63.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"officeqa-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"officeqa-pro-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-853","ingestRunId":"launch-kimi","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"OfficeQA Pro","locator":"Performance table: OfficeQA Pro","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":63.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"officeqa-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"officeqa-pro-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-854","ingestRunId":"launch-kimi","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"OfficeQA Pro","locator":"Performance table: OfficeQA Pro","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":60.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"officeqa-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"officeqa-pro-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-855","ingestRunId":"launch-kimi","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"OfficeQA Pro","locator":"Performance table: OfficeQA Pro","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":41.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"spreadsheetbench-2-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"spreadsheetbench-2-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-856","ingestRunId":"launch-kimi","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"SpreadsheetBench 2","locator":"Performance table: SpreadsheetBench 2","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":34.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"spreadsheetbench-2-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"spreadsheetbench-2-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-857","ingestRunId":"launch-kimi","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"SpreadsheetBench 2","locator":"Performance table: SpreadsheetBench 2","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":34.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"spreadsheetbench-2-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"spreadsheetbench-2-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-858","ingestRunId":"launch-kimi","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"SpreadsheetBench 2","locator":"Performance table: SpreadsheetBench 2","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":32.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"spreadsheetbench-2-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"spreadsheetbench-2-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-859","ingestRunId":"launch-kimi","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"SpreadsheetBench 2","locator":"Performance table: SpreadsheetBench 2","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":31.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"spreadsheetbench-2-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"spreadsheetbench-2-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-860","ingestRunId":"launch-kimi","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"SpreadsheetBench 2","locator":"Performance table: SpreadsheetBench 2","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":29.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"spreadsheetbench-2-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"spreadsheetbench-2-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-861","ingestRunId":"launch-kimi","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"SpreadsheetBench 2","locator":"Performance table: SpreadsheetBench 2","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":28.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"osworld-verified-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"osworld-verified-source-release-snapshot-version-not-specified-pct:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-862","ingestRunId":"launch-kimi","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"OSWorld","locator":"Performance table: OSWorld-Verified","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":84.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"osworld-verified-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"osworld-verified-source-release-snapshot-version-not-specified-pct:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-863","ingestRunId":"launch-kimi","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"OSWorld","locator":"Performance table: OSWorld-Verified","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":85,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"osworld-verified-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"osworld-verified-source-release-snapshot-version-not-specified-pct:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-864","ingestRunId":"launch-kimi","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"OSWorld","locator":"Performance table: OSWorld-Verified","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":83,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"osworld-verified-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"osworld-verified-source-release-snapshot-version-not-specified-pct:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-865","ingestRunId":"launch-kimi","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"OSWorld","locator":"Performance table: OSWorld-Verified","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":83.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"osworld-verified-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"osworld-verified-source-release-snapshot-version-not-specified-pct:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-866","ingestRunId":"launch-kimi","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"OSWorld","locator":"Performance table: OSWorld-Verified","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":79,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"osworld-2-0","evidenceKind":"lab_self_report","harnessId":"osworld-2-0:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-867","ingestRunId":"launch-kimi","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"OSWorld","locator":"Performance table: OSWorld 2.0","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":58.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"osworld-2-0","evidenceKind":"lab_self_report","harnessId":"osworld-2-0:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-868","ingestRunId":"launch-kimi","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"OSWorld","locator":"Performance table: OSWorld 2.0","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":66.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"osworld-2-0","evidenceKind":"lab_self_report","harnessId":"osworld-2-0:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-869","ingestRunId":"launch-kimi","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"OSWorld","locator":"Performance table: OSWorld 2.0","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":62.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"osworld-2-0","evidenceKind":"lab_self_report","harnessId":"osworld-2-0:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-870","ingestRunId":"launch-kimi","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"OSWorld","locator":"Performance table: OSWorld 2.0","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":55.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"osworld-2-0","evidenceKind":"lab_self_report","harnessId":"osworld-2-0:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-871","ingestRunId":"launch-kimi","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"OSWorld","locator":"Performance table: OSWorld 2.0","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":49.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"saas-bench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"saas-bench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-872","ingestRunId":"launch-kimi","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"SaaS-Bench","locator":"Performance table: SaaS-Bench","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":60.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"saas-bench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"saas-bench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-873","ingestRunId":"launch-kimi","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"SaaS-Bench","locator":"Performance table: SaaS-Bench","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":61.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"saas-bench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"saas-bench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-874","ingestRunId":"launch-kimi","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"SaaS-Bench","locator":"Performance table: SaaS-Bench","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":56.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"saas-bench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"saas-bench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-875","ingestRunId":"launch-kimi","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"SaaS-Bench","locator":"Performance table: SaaS-Bench","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":43.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"tau3-banking-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"tau3-banking-source-release-snapshot-version-not-specified-pct:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-876","ingestRunId":"launch-kimi","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites Artificial Analysis; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"τ³-Banking","locator":"Performance table: τ³-Banking","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":33.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"tau3-banking-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"tau3-banking-source-release-snapshot-version-not-specified-pct:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-877","ingestRunId":"launch-kimi","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites Artificial Analysis; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"τ³-Banking","locator":"Performance table: τ³-Banking","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":26.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"tau3-banking-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"tau3-banking-source-release-snapshot-version-not-specified-pct:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-878","ingestRunId":"launch-kimi","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites Artificial Analysis; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"τ³-Banking","locator":"Performance table: τ³-Banking","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":33,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"tau3-banking-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"tau3-banking-source-release-snapshot-version-not-specified-pct:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-879","ingestRunId":"launch-kimi","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites Artificial Analysis; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"τ³-Banking","locator":"Performance table: τ³-Banking","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":27.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"tau3-banking-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"tau3-banking-source-release-snapshot-version-not-specified-pct:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-880","ingestRunId":"launch-kimi","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites Artificial Analysis; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"τ³-Banking","locator":"Performance table: τ³-Banking","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":31.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"tau3-banking-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"tau3-banking-source-release-snapshot-version-not-specified-pct:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-881","ingestRunId":"launch-kimi","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites Artificial Analysis; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"τ³-Banking","locator":"Performance table: τ³-Banking","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":26.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"harvey-lab-aa-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"harvey-lab-aa-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-882","ingestRunId":"launch-kimi","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites Artificial Analysis; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Harvey Lab-AA","locator":"Performance table: Harvey Lab-AA","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":94.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"harvey-lab-aa-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"harvey-lab-aa-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-883","ingestRunId":"launch-kimi","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites Artificial Analysis; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Harvey Lab-AA","locator":"Performance table: Harvey Lab-AA","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":93.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"harvey-lab-aa-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"harvey-lab-aa-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-884","ingestRunId":"launch-kimi","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites Artificial Analysis; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Harvey Lab-AA","locator":"Performance table: Harvey Lab-AA","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":87.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"harvey-lab-aa-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"harvey-lab-aa-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-885","ingestRunId":"launch-kimi","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites Artificial Analysis; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Harvey Lab-AA","locator":"Performance table: Harvey Lab-AA","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":91.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"harvey-lab-aa-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"harvey-lab-aa-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-886","ingestRunId":"launch-kimi","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites Artificial Analysis; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Harvey Lab-AA","locator":"Performance table: Harvey Lab-AA","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":86.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"harvey-lab-aa-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"harvey-lab-aa-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-887","ingestRunId":"launch-kimi","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites Artificial Analysis; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Harvey Lab-AA","locator":"Performance table: Harvey Lab-AA","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":91,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"corpfin-v2-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"corpfin-v2-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-888","ingestRunId":"launch-kimi","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites Vals AI; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"CorpFin v2","locator":"Performance table: CorpFin v2","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":71.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"corpfin-v2-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"corpfin-v2-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-889","ingestRunId":"launch-kimi","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites Vals AI; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"CorpFin v2","locator":"Performance table: CorpFin v2","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":71.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"corpfin-v2-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"corpfin-v2-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-890","ingestRunId":"launch-kimi","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites Vals AI; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"CorpFin v2","locator":"Performance table: CorpFin v2","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":64.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"corpfin-v2-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"corpfin-v2-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-891","ingestRunId":"launch-kimi","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites Vals AI; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"CorpFin v2","locator":"Performance table: CorpFin v2","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":66.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"corpfin-v2-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"corpfin-v2-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-892","ingestRunId":"launch-kimi","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites Vals AI; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"CorpFin v2","locator":"Performance table: CorpFin v2","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":68.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"corpfin-v2-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"corpfin-v2-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-893","ingestRunId":"launch-kimi","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites Vals AI; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"CorpFin v2","locator":"Performance table: CorpFin v2","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":66.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"finance-agent-v2-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"finance-agent-v2-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-894","ingestRunId":"launch-kimi","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites Vals AI; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Finance Agent v2","locator":"Performance table: Finance Agent v2","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":54.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"finance-agent-v2-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"finance-agent-v2-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-895","ingestRunId":"launch-kimi","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites Vals AI; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Finance Agent v2","locator":"Performance table: Finance Agent v2","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":56.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"finance-agent-v2-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"finance-agent-v2-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-896","ingestRunId":"launch-kimi","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites Vals AI; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Finance Agent v2","locator":"Performance table: Finance Agent v2","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":53.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"finance-agent-v2-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"finance-agent-v2-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-897","ingestRunId":"launch-kimi","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites Vals AI; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Finance Agent v2","locator":"Performance table: Finance Agent v2","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":53.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"finance-agent-v2-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"finance-agent-v2-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-898","ingestRunId":"launch-kimi","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites Vals AI; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Finance Agent v2","locator":"Performance table: Finance Agent v2","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":51.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"finance-agent-v2-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"finance-agent-v2-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-899","ingestRunId":"launch-kimi","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites Vals AI; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Finance Agent v2","locator":"Performance table: Finance Agent v2","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":49.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"legal-research-bench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"legal-research-bench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-900","ingestRunId":"launch-kimi","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites Vals AI; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Legal Research Bench","locator":"Performance table: Legal Research Bench","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":44.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"legal-research-bench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"legal-research-bench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-901","ingestRunId":"launch-kimi","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites Vals AI; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Legal Research Bench","locator":"Performance table: Legal Research Bench","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":49.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"legal-research-bench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"legal-research-bench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-902","ingestRunId":"launch-kimi","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites Vals AI; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Legal Research Bench","locator":"Performance table: Legal Research Bench","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":48.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"legal-research-bench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"legal-research-bench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-903","ingestRunId":"launch-kimi","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites Vals AI; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Legal Research Bench","locator":"Performance table: Legal Research Bench","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":43.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"legal-research-bench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"legal-research-bench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-904","ingestRunId":"launch-kimi","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites Vals AI; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Legal Research Bench","locator":"Performance table: Legal Research Bench","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":40.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"legal-research-bench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"legal-research-bench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-905","ingestRunId":"launch-kimi","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","comparabilityReason":"Kimi K3 Evaluation Details cites Vals AI; retain as published context, not a new Moonshot comparison.","comparable":false,"configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Legal Research Bench","locator":"Performance table: Legal Research Bench","metric":"%","notes":"Cited result; see benchmark-specific Evaluation Details.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":31.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"worldvqa-forceanswer-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"worldvqa-forceanswer-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-906","ingestRunId":"launch-kimi","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"WorldVQA ForceAnswer","locator":"Performance table: WorldVQA ForceAnswer","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":51,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"worldvqa-forceanswer-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"worldvqa-forceanswer-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-907","ingestRunId":"launch-kimi","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"WorldVQA ForceAnswer","locator":"Performance table: WorldVQA ForceAnswer","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":56.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"worldvqa-forceanswer-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"worldvqa-forceanswer-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-908","ingestRunId":"launch-kimi","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"WorldVQA ForceAnswer","locator":"Performance table: WorldVQA ForceAnswer","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":41.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"worldvqa-forceanswer-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"worldvqa-forceanswer-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-909","ingestRunId":"launch-kimi","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"WorldVQA ForceAnswer","locator":"Performance table: WorldVQA ForceAnswer","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":39.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"worldvqa-forceanswer-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"worldvqa-forceanswer-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-910","ingestRunId":"launch-kimi","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"coding","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"WorldVQA ForceAnswer","locator":"Performance table: WorldVQA ForceAnswer","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":38.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"omnidocbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"omnidocbench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-911","ingestRunId":"launch-kimi","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"multimodal","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"OmniDocBench","locator":"Performance table: OmniDocBench","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":91.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"omnidocbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"omnidocbench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-912","ingestRunId":"launch-kimi","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"multimodal","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"OmniDocBench","locator":"Performance table: OmniDocBench","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":89.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"omnidocbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"omnidocbench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-913","ingestRunId":"launch-kimi","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"multimodal","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"OmniDocBench","locator":"Performance table: OmniDocBench","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":85.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"omnidocbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"omnidocbench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-914","ingestRunId":"launch-kimi","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"multimodal","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"OmniDocBench","locator":"Performance table: OmniDocBench","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":87.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"omnidocbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"omnidocbench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-915","ingestRunId":"launch-kimi","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"multimodal","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"OmniDocBench","locator":"Performance table: OmniDocBench","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":89.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"perceptionbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"perceptionbench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-916","ingestRunId":"launch-kimi","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"multimodal","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"PerceptionBench","locator":"Performance table: PerceptionBench","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":58.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"perceptionbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"perceptionbench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-917","ingestRunId":"launch-kimi","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"multimodal","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"PerceptionBench","locator":"Performance table: PerceptionBench","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":57.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"perceptionbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"perceptionbench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-918","ingestRunId":"launch-kimi","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"multimodal","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"PerceptionBench","locator":"Performance table: PerceptionBench","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":59.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"perceptionbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"perceptionbench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-919","ingestRunId":"launch-kimi","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"multimodal","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"PerceptionBench","locator":"Performance table: PerceptionBench","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":47.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"perceptionbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"perceptionbench-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-920","ingestRunId":"launch-kimi","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"multimodal","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"PerceptionBench","locator":"Performance table: PerceptionBench","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":55.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"video-mme-w-sub-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"video-mme-w-sub-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-921","ingestRunId":"launch-kimi","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"multimodal","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Video-MME (w. sub)","locator":"Performance table: Video-MME (w. sub)","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":90,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"video-mme-w-sub-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"video-mme-w-sub-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-922","ingestRunId":"launch-kimi","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"multimodal","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Video-MME (w. sub)","locator":"Performance table: Video-MME (w. sub)","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":89.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"video-mme-w-sub-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"video-mme-w-sub-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-923","ingestRunId":"launch-kimi","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"multimodal","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Video-MME (w. sub)","locator":"Performance table: Video-MME (w. sub)","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":86,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"video-mme-w-sub-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"video-mme-w-sub-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-924","ingestRunId":"launch-kimi","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"multimodal","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"Video-MME (w. sub)","locator":"Performance table: Video-MME (w. sub)","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":89.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"mmvu-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mmvu-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-925","ingestRunId":"launch-kimi","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"multimodal","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"MMVU","locator":"Performance table: MMVU","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":82.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"mmvu-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mmvu-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-926","ingestRunId":"launch-kimi","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"multimodal","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"MMVU","locator":"Performance table: MMVU","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":81.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"mmvu-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mmvu-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-927","ingestRunId":"launch-kimi","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"multimodal","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"MMVU","locator":"Performance table: MMVU","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":79.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"mmvu-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mmvu-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-928","ingestRunId":"launch-kimi","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"multimodal","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"MMVU","locator":"Performance table: MMVU","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":81.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"babyvision-w-python-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"babyvision-w-python-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-929","ingestRunId":"launch-kimi","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"multimodal","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"BabyVision w/ python","locator":"Performance table: BabyVision w/ python","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":85.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"babyvision-w-python-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"babyvision-w-python-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-930","ingestRunId":"launch-kimi","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"multimodal","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"BabyVision w/ python","locator":"Performance table: BabyVision w/ python","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":90.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"babyvision-w-python-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"babyvision-w-python-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-931","ingestRunId":"launch-kimi","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"multimodal","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"BabyVision w/ python","locator":"Performance table: BabyVision w/ python","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":88.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"babyvision-w-python-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"babyvision-w-python-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-932","ingestRunId":"launch-kimi","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"multimodal","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"BabyVision w/ python","locator":"Performance table: BabyVision w/ python","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":81.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"babyvision-w-python-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"babyvision-w-python-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gS2ltaSB0ZW1wMTsgc2luZ2xlLXN0ZXAgdG9wX3AuOTUsYWdlbnRpYyB0b3BfcDE7IHZpc2lvbiB0b29scz1QeXRob24sM3J1bnMgZXhjZXB0IFplcm9CZW5jaDVydW5zLg","id":"launch-kimi-933","ingestRunId":"launch-kimi","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"multimodal","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"BabyVision w/ python","locator":"Performance table: BabyVision w/ python","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":83.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"mmmu-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mmmu-pro-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aG91dCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","id":"launch-kimi-934","ingestRunId":"launch-kimi","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"multimodal","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"MMMU-Pro","locator":"Performance table: MMMU-Pro","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":81.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"mmmu-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mmmu-pro-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","id":"launch-kimi-935","ingestRunId":"launch-kimi","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"multimodal","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"MMMU-Pro","locator":"Performance table: MMMU-Pro","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":83.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"mmmu-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mmmu-pro-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aG91dCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","id":"launch-kimi-936","ingestRunId":"launch-kimi","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"multimodal","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"MMMU-Pro","locator":"Performance table: MMMU-Pro","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":81.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"mmmu-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mmmu-pro-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","id":"launch-kimi-937","ingestRunId":"launch-kimi","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"multimodal","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"MMMU-Pro","locator":"Performance table: MMMU-Pro","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":86.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"mmmu-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mmmu-pro-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aG91dCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","id":"launch-kimi-938","ingestRunId":"launch-kimi","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"multimodal","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"MMMU-Pro","locator":"Performance table: MMMU-Pro","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":83,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"mmmu-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mmmu-pro-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","id":"launch-kimi-939","ingestRunId":"launch-kimi","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"multimodal","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"MMMU-Pro","locator":"Performance table: MMMU-Pro","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":84.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"mmmu-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mmmu-pro-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aG91dCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","id":"launch-kimi-940","ingestRunId":"launch-kimi","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"multimodal","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"MMMU-Pro","locator":"Performance table: MMMU-Pro","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":78.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"mmmu-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mmmu-pro-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","id":"launch-kimi-941","ingestRunId":"launch-kimi","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"multimodal","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"MMMU-Pro","locator":"Performance table: MMMU-Pro","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":82.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"mmmu-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mmmu-pro-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aG91dCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","id":"launch-kimi-942","ingestRunId":"launch-kimi","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"multimodal","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"MMMU-Pro","locator":"Performance table: MMMU-Pro","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":81.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"mmmu-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mmmu-pro-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","id":"launch-kimi-943","ingestRunId":"launch-kimi","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"multimodal","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"MMMU-Pro","locator":"Performance table: MMMU-Pro","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":83.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"charxiv-rq-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"charxiv-rq-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aG91dCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","id":"launch-kimi-944","ingestRunId":"launch-kimi","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"multimodal","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"CharXiv (RQ)","locator":"Performance table: CharXiv (RQ)","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":84.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"charxiv-rq-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"charxiv-rq-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","id":"launch-kimi-945","ingestRunId":"launch-kimi","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"multimodal","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"CharXiv (RQ)","locator":"Performance table: CharXiv (RQ)","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":91.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"charxiv-rq-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"charxiv-rq-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aG91dCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","id":"launch-kimi-946","ingestRunId":"launch-kimi","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"multimodal","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"CharXiv (RQ)","locator":"Performance table: CharXiv (RQ)","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":88.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"charxiv-rq-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"charxiv-rq-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","id":"launch-kimi-947","ingestRunId":"launch-kimi","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"multimodal","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"CharXiv (RQ)","locator":"Performance table: CharXiv (RQ)","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":93.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"charxiv-rq-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"charxiv-rq-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aG91dCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","id":"launch-kimi-948","ingestRunId":"launch-kimi","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"multimodal","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"CharXiv (RQ)","locator":"Performance table: CharXiv (RQ)","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":84.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"charxiv-rq-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"charxiv-rq-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","id":"launch-kimi-949","ingestRunId":"launch-kimi","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"multimodal","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"CharXiv (RQ)","locator":"Performance table: CharXiv (RQ)","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":89.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"charxiv-rq-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"charxiv-rq-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aG91dCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","id":"launch-kimi-950","ingestRunId":"launch-kimi","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"multimodal","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"CharXiv (RQ)","locator":"Performance table: CharXiv (RQ)","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":80.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"charxiv-rq-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"charxiv-rq-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","id":"launch-kimi-951","ingestRunId":"launch-kimi","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"multimodal","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"CharXiv (RQ)","locator":"Performance table: CharXiv (RQ)","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":89.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"charxiv-rq-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"charxiv-rq-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aG91dCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","id":"launch-kimi-952","ingestRunId":"launch-kimi","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"multimodal","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"CharXiv (RQ)","locator":"Performance table: CharXiv (RQ)","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":84.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"charxiv-rq-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"charxiv-rq-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","id":"launch-kimi-953","ingestRunId":"launch-kimi","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"multimodal","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"CharXiv (RQ)","locator":"Performance table: CharXiv (RQ)","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":89,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"mathvision-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mathvision-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aG91dCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","id":"launch-kimi-954","ingestRunId":"launch-kimi","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"multimodal","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"MathVision","locator":"Performance table: MathVision","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":94.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"mathvision-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mathvision-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","id":"launch-kimi-955","ingestRunId":"launch-kimi","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"multimodal","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"MathVision","locator":"Performance table: MathVision","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":97.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"mathvision-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mathvision-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aG91dCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","id":"launch-kimi-956","ingestRunId":"launch-kimi","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"multimodal","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"MathVision","locator":"Performance table: MathVision","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":94.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"mathvision-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mathvision-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","id":"launch-kimi-957","ingestRunId":"launch-kimi","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"multimodal","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"MathVision","locator":"Performance table: MathVision","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":98.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"mathvision-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mathvision-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aG91dCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","id":"launch-kimi-958","ingestRunId":"launch-kimi","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"multimodal","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"MathVision","locator":"Performance table: MathVision","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":95.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"mathvision-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mathvision-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","id":"launch-kimi-959","ingestRunId":"launch-kimi","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"multimodal","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"MathVision","locator":"Performance table: MathVision","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":97.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"mathvision-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mathvision-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aG91dCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","id":"launch-kimi-960","ingestRunId":"launch-kimi","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"multimodal","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"MathVision","locator":"Performance table: MathVision","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":86.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"mathvision-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mathvision-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","id":"launch-kimi-961","ingestRunId":"launch-kimi","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"multimodal","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"MathVision","locator":"Performance table: MathVision","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":97.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"mathvision-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mathvision-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aG91dCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","id":"launch-kimi-962","ingestRunId":"launch-kimi","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"multimodal","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"MathVision","locator":"Performance table: MathVision","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":92.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"mathvision-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mathvision-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","id":"launch-kimi-963","ingestRunId":"launch-kimi","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"multimodal","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"MathVision","locator":"Performance table: MathVision","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":96.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"zerobench-pass-5-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"zerobench-pass-5-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aG91dCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","id":"launch-kimi-964","ingestRunId":"launch-kimi","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"ZeroBench (pass@5)","locator":"Performance table: ZeroBench (pass@5)","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":23,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"zerobench-pass-5-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"zerobench-pass-5-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","id":"launch-kimi-965","ingestRunId":"launch-kimi","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"ZeroBench (pass@5)","locator":"Performance table: ZeroBench (pass@5)","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":41,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"zerobench-pass-5-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"zerobench-pass-5-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aG91dCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","id":"launch-kimi-966","ingestRunId":"launch-kimi","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"ZeroBench (pass@5)","locator":"Performance table: ZeroBench (pass@5)","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":23,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"zerobench-pass-5-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"zerobench-pass-5-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","id":"launch-kimi-967","ingestRunId":"launch-kimi","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"ZeroBench (pass@5)","locator":"Performance table: ZeroBench (pass@5)","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":46,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"zerobench-pass-5-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"zerobench-pass-5-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aG91dCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","id":"launch-kimi-968","ingestRunId":"launch-kimi","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"ZeroBench (pass@5)","locator":"Performance table: ZeroBench (pass@5)","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":17,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"zerobench-pass-5-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"zerobench-pass-5-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","id":"launch-kimi-969","ingestRunId":"launch-kimi","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"ZeroBench (pass@5)","locator":"Performance table: ZeroBench (pass@5)","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":35,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"zerobench-pass-5-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"zerobench-pass-5-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aG91dCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","id":"launch-kimi-970","ingestRunId":"launch-kimi","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"ZeroBench (pass@5)","locator":"Performance table: ZeroBench (pass@5)","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":17,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"zerobench-pass-5-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"zerobench-pass-5-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","id":"launch-kimi-971","ingestRunId":"launch-kimi","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"ZeroBench (pass@5)","locator":"Performance table: ZeroBench (pass@5)","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":34,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"zerobench-pass-5-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"zerobench-pass-5-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aG91dCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","id":"launch-kimi-972","ingestRunId":"launch-kimi","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"ZeroBench (pass@5)","locator":"Performance table: ZeroBench (pass@5)","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":22,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"zerobench-pass-5-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"zerobench-pass-5-source-release-snapshot-version-not-specified:kimi:VGFibGUgaGVhZGVyczogS2ltaSBLMywgR1BULTUuNiBTb2wsIE9wdXMgNC44IGFuZCBHTE0tNS4yIG1heDsgR1BULTUuNSB4aGlnaDsgRmFibGUgNSBtYXggd2l0aCBmYWxsYmFja3MuIEJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMgb3ZlcnJpZGUgaGVhZGVyIHNldHRpbmdzOyBjaXRlZCByZXN1bHRzIGFyZSBub3QgbmV3IE1vb25zaG90IGV2YWx1YXRpb25zLiBCZW5jaG1hcmstc3BlY2lmaWMgdG9vbHMsIGhhcm5lc3MgYW5kIGJ1ZGdldCBkb2N1bWVudGVkIHVuZGVyIEV2YWx1YXRpb24gRGV0YWlscy4gd2l0aCB0b29scyBLaW1pIHRlbXAxOyBzaW5nbGUtc3RlcCB0b3BfcC45NSxhZ2VudGljIHRvcF9wMTsgdmlzaW9uIHRvb2xzPVB5dGhvbiwzcnVucyBleGNlcHQgWmVyb0JlbmNoNXJ1bnMu","id":"launch-kimi-973","ingestRunId":"launch-kimi","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"57de265b5842dfa465c6e73b368b0e15a89b8793b5450528dad577da202cc6fe","report":{"category":"agentic","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs.","family":"ZeroBench (pass@5)","locator":"Performance table: ZeroBench (pass@5)","metric":"%","notes":"First-party reported result.","sourceId":"kimi","sourceTitle":"moonshotai/Kimi-K3"},"score":41,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md"},{"benchmarkId":"hle-full-w-tools","evidenceKind":"lab_self_report","harnessId":"hle-full-w-tools:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEhMRS1GdWxsICh3LyB0b29scykuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMu","id":"launch-kimi-k26-card-2106","ingestRunId":"launch-kimi-k26-card","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"agentic","comparable":true,"configuration":"Kimi K2.6 card, HLE-Full (w/ tools). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons.","family":"Humanity's Last Exam","locator":"Evaluation Results table, HLE-Full (w/ tools), column 2","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":54,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"hle-full-w-tools","evidenceKind":"lab_self_report","harnessId":"hle-full-w-tools:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEhMRS1GdWxsICh3LyB0b29scykuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMu","id":"launch-kimi-k26-card-2107","ingestRunId":"launch-kimi-k26-card","modelId":"gpt-5.4","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"agentic","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, HLE-Full (w/ tools). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons.","family":"Humanity's Last Exam","locator":"Evaluation Results table, HLE-Full (w/ tools), column 3","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":52.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"hle-full-w-tools","evidenceKind":"lab_self_report","harnessId":"hle-full-w-tools:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEhMRS1GdWxsICh3LyB0b29scykuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMu","id":"launch-kimi-k26-card-2108","ingestRunId":"launch-kimi-k26-card","modelId":"claude-opus-4-6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"agentic","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, HLE-Full (w/ tools). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons.","family":"Humanity's Last Exam","locator":"Evaluation Results table, HLE-Full (w/ tools), column 4","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":53,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"hle-full-w-tools","evidenceKind":"lab_self_report","harnessId":"hle-full-w-tools:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEhMRS1GdWxsICh3LyB0b29scykuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMu","id":"launch-kimi-k26-card-2109","ingestRunId":"launch-kimi-k26-card","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"agentic","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, HLE-Full (w/ tools). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons.","family":"Humanity's Last Exam","locator":"Evaluation Results table, HLE-Full (w/ tools), column 5","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":51.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"browsecomp-percent","evidenceKind":"lab_self_report","harnessId":"browsecomp-percent:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEJyb3dzZUNvbXAuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMu","id":"launch-kimi-k26-card-2110","ingestRunId":"launch-kimi-k26-card","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"agentic","comparable":true,"configuration":"Kimi K2.6 card, BrowseComp. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons.","family":"BrowseComp","locator":"Evaluation Results table, BrowseComp, column 2","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":83.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"browsecomp-percent","evidenceKind":"lab_self_report","harnessId":"browsecomp-percent:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEJyb3dzZUNvbXAuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMu","id":"launch-kimi-k26-card-2111","ingestRunId":"launch-kimi-k26-card","modelId":"gpt-5.4","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"agentic","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, BrowseComp. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons.","family":"BrowseComp","locator":"Evaluation Results table, BrowseComp, column 3","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":82.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"browsecomp-percent","evidenceKind":"lab_self_report","harnessId":"browsecomp-percent:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEJyb3dzZUNvbXAuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMu","id":"launch-kimi-k26-card-2112","ingestRunId":"launch-kimi-k26-card","modelId":"claude-opus-4-6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"agentic","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, BrowseComp. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons.","family":"BrowseComp","locator":"Evaluation Results table, BrowseComp, column 4","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":83.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"browsecomp-percent","evidenceKind":"lab_self_report","harnessId":"browsecomp-percent:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEJyb3dzZUNvbXAuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMu","id":"launch-kimi-k26-card-2113","ingestRunId":"launch-kimi-k26-card","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"agentic","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, BrowseComp. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons.","family":"BrowseComp","locator":"Evaluation Results table, BrowseComp, column 5","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":85.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"browsecomp-agent-swarm","evidenceKind":"lab_self_report","harnessId":"browsecomp-agent-swarm:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEJyb3dzZUNvbXAgKEFnZW50IFN3YXJtKS4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4","id":"launch-kimi-k26-card-2114","ingestRunId":"launch-kimi-k26-card","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"agentic","comparabilityReason":"Agent swarm execution is not an individual model configuration.","comparable":false,"configuration":"Kimi K2.6 card, BrowseComp (Agent Swarm). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons.","family":"BrowseComp","locator":"Evaluation Results table, BrowseComp (Agent Swarm), column 2","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":86.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"browsecomp-agent-swarm","evidenceKind":"lab_self_report","harnessId":"browsecomp-agent-swarm:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEJyb3dzZUNvbXAgKEFnZW50IFN3YXJtKS4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4","id":"launch-kimi-k26-card-2115","ingestRunId":"launch-kimi-k26-card","modelId":"gpt-5.4","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"agentic","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, BrowseComp (Agent Swarm). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons.","family":"BrowseComp","locator":"Evaluation Results table, BrowseComp (Agent Swarm), column 3","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":82.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"browsecomp-agent-swarm","evidenceKind":"lab_self_report","harnessId":"browsecomp-agent-swarm:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEJyb3dzZUNvbXAgKEFnZW50IFN3YXJtKS4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4","id":"launch-kimi-k26-card-2116","ingestRunId":"launch-kimi-k26-card","modelId":"claude-opus-4-6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"agentic","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, BrowseComp (Agent Swarm). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons.","family":"BrowseComp","locator":"Evaluation Results table, BrowseComp (Agent Swarm), column 4","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":83.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"browsecomp-agent-swarm","evidenceKind":"lab_self_report","harnessId":"browsecomp-agent-swarm:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEJyb3dzZUNvbXAgKEFnZW50IFN3YXJtKS4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4","id":"launch-kimi-k26-card-2117","ingestRunId":"launch-kimi-k26-card","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"agentic","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, BrowseComp (Agent Swarm). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons.","family":"BrowseComp","locator":"Evaluation Results table, BrowseComp (Agent Swarm), column 5","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":85.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"deepsearchqa-f1-score","evidenceKind":"lab_self_report","harnessId":"deepsearchqa-f1-score:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIERlZXBTZWFyY2hRQSAoZjEtc2NvcmUpLiBUaGlua2luZyBtb2RlOyB0ZW1wZXJhdHVyZSAxLjAsIHRvcC1wIDEuMCwgY29udGV4dCAyNjIxNDQuIENvbXBhcmF0b3Igc2V0dGluZ3M6IEdQVC01LjQgeGhpZ2gsIE9wdXM0LjYgbWF4LCBHZW1pbmkzLjFQcm8gaGlnaC4gU2VlIGJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMuIE9ubHkgb3duIEsyLjYgYW5kIHN0YXJyZWQgcmUtZXZhbHVhdGlvbnMgY2FuIGZvcm0gbmV3IGNvbXBhcmlzb25zLg","id":"launch-kimi-k26-card-2118","ingestRunId":"launch-kimi-k26-card","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"agentic","comparable":true,"configuration":"Kimi K2.6 card, DeepSearchQA (f1-score). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons.","family":"DeepSearchQA (f1-score)","locator":"Evaluation Results table, DeepSearchQA (f1-score), column 2","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":92.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"deepsearchqa-f1-score","evidenceKind":"lab_self_report","harnessId":"deepsearchqa-f1-score:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIERlZXBTZWFyY2hRQSAoZjEtc2NvcmUpLiBUaGlua2luZyBtb2RlOyB0ZW1wZXJhdHVyZSAxLjAsIHRvcC1wIDEuMCwgY29udGV4dCAyNjIxNDQuIENvbXBhcmF0b3Igc2V0dGluZ3M6IEdQVC01LjQgeGhpZ2gsIE9wdXM0LjYgbWF4LCBHZW1pbmkzLjFQcm8gaGlnaC4gU2VlIGJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMuIE9ubHkgb3duIEsyLjYgYW5kIHN0YXJyZWQgcmUtZXZhbHVhdGlvbnMgY2FuIGZvcm0gbmV3IGNvbXBhcmlzb25zLg","id":"launch-kimi-k26-card-2119","ingestRunId":"launch-kimi-k26-card","modelId":"gpt-5.4","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"agentic","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, DeepSearchQA (f1-score). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons.","family":"DeepSearchQA (f1-score)","locator":"Evaluation Results table, DeepSearchQA (f1-score), column 3","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":78.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"deepsearchqa-f1-score","evidenceKind":"lab_self_report","harnessId":"deepsearchqa-f1-score:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIERlZXBTZWFyY2hRQSAoZjEtc2NvcmUpLiBUaGlua2luZyBtb2RlOyB0ZW1wZXJhdHVyZSAxLjAsIHRvcC1wIDEuMCwgY29udGV4dCAyNjIxNDQuIENvbXBhcmF0b3Igc2V0dGluZ3M6IEdQVC01LjQgeGhpZ2gsIE9wdXM0LjYgbWF4LCBHZW1pbmkzLjFQcm8gaGlnaC4gU2VlIGJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMuIE9ubHkgb3duIEsyLjYgYW5kIHN0YXJyZWQgcmUtZXZhbHVhdGlvbnMgY2FuIGZvcm0gbmV3IGNvbXBhcmlzb25zLg","id":"launch-kimi-k26-card-2120","ingestRunId":"launch-kimi-k26-card","modelId":"claude-opus-4-6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"agentic","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, DeepSearchQA (f1-score). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons.","family":"DeepSearchQA (f1-score)","locator":"Evaluation Results table, DeepSearchQA (f1-score), column 4","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":91.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"deepsearchqa-f1-score","evidenceKind":"lab_self_report","harnessId":"deepsearchqa-f1-score:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIERlZXBTZWFyY2hRQSAoZjEtc2NvcmUpLiBUaGlua2luZyBtb2RlOyB0ZW1wZXJhdHVyZSAxLjAsIHRvcC1wIDEuMCwgY29udGV4dCAyNjIxNDQuIENvbXBhcmF0b3Igc2V0dGluZ3M6IEdQVC01LjQgeGhpZ2gsIE9wdXM0LjYgbWF4LCBHZW1pbmkzLjFQcm8gaGlnaC4gU2VlIGJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMuIE9ubHkgb3duIEsyLjYgYW5kIHN0YXJyZWQgcmUtZXZhbHVhdGlvbnMgY2FuIGZvcm0gbmV3IGNvbXBhcmlzb25zLg","id":"launch-kimi-k26-card-2121","ingestRunId":"launch-kimi-k26-card","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"agentic","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, DeepSearchQA (f1-score). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons.","family":"DeepSearchQA (f1-score)","locator":"Evaluation Results table, DeepSearchQA (f1-score), column 5","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":81.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"deepsearchqa-accuracy","evidenceKind":"lab_self_report","harnessId":"deepsearchqa-accuracy:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIERlZXBTZWFyY2hRQSAoYWNjdXJhY3kpLiBUaGlua2luZyBtb2RlOyB0ZW1wZXJhdHVyZSAxLjAsIHRvcC1wIDEuMCwgY29udGV4dCAyNjIxNDQuIENvbXBhcmF0b3Igc2V0dGluZ3M6IEdQVC01LjQgeGhpZ2gsIE9wdXM0LjYgbWF4LCBHZW1pbmkzLjFQcm8gaGlnaC4gU2VlIGJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMuIE9ubHkgb3duIEsyLjYgYW5kIHN0YXJyZWQgcmUtZXZhbHVhdGlvbnMgY2FuIGZvcm0gbmV3IGNvbXBhcmlzb25zLg","id":"launch-kimi-k26-card-2122","ingestRunId":"launch-kimi-k26-card","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"agentic","comparable":true,"configuration":"Kimi K2.6 card, DeepSearchQA (accuracy). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons.","family":"DeepSearchQA (accuracy)","locator":"Evaluation Results table, DeepSearchQA (accuracy), column 2","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":83,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"deepsearchqa-accuracy","evidenceKind":"lab_self_report","harnessId":"deepsearchqa-accuracy:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIERlZXBTZWFyY2hRQSAoYWNjdXJhY3kpLiBUaGlua2luZyBtb2RlOyB0ZW1wZXJhdHVyZSAxLjAsIHRvcC1wIDEuMCwgY29udGV4dCAyNjIxNDQuIENvbXBhcmF0b3Igc2V0dGluZ3M6IEdQVC01LjQgeGhpZ2gsIE9wdXM0LjYgbWF4LCBHZW1pbmkzLjFQcm8gaGlnaC4gU2VlIGJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMuIE9ubHkgb3duIEsyLjYgYW5kIHN0YXJyZWQgcmUtZXZhbHVhdGlvbnMgY2FuIGZvcm0gbmV3IGNvbXBhcmlzb25zLg","id":"launch-kimi-k26-card-2123","ingestRunId":"launch-kimi-k26-card","modelId":"gpt-5.4","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"agentic","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, DeepSearchQA (accuracy). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons.","family":"DeepSearchQA (accuracy)","locator":"Evaluation Results table, DeepSearchQA (accuracy), column 3","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":63.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"deepsearchqa-accuracy","evidenceKind":"lab_self_report","harnessId":"deepsearchqa-accuracy:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIERlZXBTZWFyY2hRQSAoYWNjdXJhY3kpLiBUaGlua2luZyBtb2RlOyB0ZW1wZXJhdHVyZSAxLjAsIHRvcC1wIDEuMCwgY29udGV4dCAyNjIxNDQuIENvbXBhcmF0b3Igc2V0dGluZ3M6IEdQVC01LjQgeGhpZ2gsIE9wdXM0LjYgbWF4LCBHZW1pbmkzLjFQcm8gaGlnaC4gU2VlIGJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMuIE9ubHkgb3duIEsyLjYgYW5kIHN0YXJyZWQgcmUtZXZhbHVhdGlvbnMgY2FuIGZvcm0gbmV3IGNvbXBhcmlzb25zLg","id":"launch-kimi-k26-card-2124","ingestRunId":"launch-kimi-k26-card","modelId":"claude-opus-4-6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"agentic","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, DeepSearchQA (accuracy). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons.","family":"DeepSearchQA (accuracy)","locator":"Evaluation Results table, DeepSearchQA (accuracy), column 4","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":80.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"deepsearchqa-accuracy","evidenceKind":"lab_self_report","harnessId":"deepsearchqa-accuracy:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIERlZXBTZWFyY2hRQSAoYWNjdXJhY3kpLiBUaGlua2luZyBtb2RlOyB0ZW1wZXJhdHVyZSAxLjAsIHRvcC1wIDEuMCwgY29udGV4dCAyNjIxNDQuIENvbXBhcmF0b3Igc2V0dGluZ3M6IEdQVC01LjQgeGhpZ2gsIE9wdXM0LjYgbWF4LCBHZW1pbmkzLjFQcm8gaGlnaC4gU2VlIGJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMuIE9ubHkgb3duIEsyLjYgYW5kIHN0YXJyZWQgcmUtZXZhbHVhdGlvbnMgY2FuIGZvcm0gbmV3IGNvbXBhcmlzb25zLg","id":"launch-kimi-k26-card-2125","ingestRunId":"launch-kimi-k26-card","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"agentic","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, DeepSearchQA (accuracy). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons.","family":"DeepSearchQA (accuracy)","locator":"Evaluation Results table, DeepSearchQA (accuracy), column 5","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":60.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"widesearch-item-f1","evidenceKind":"lab_self_report","harnessId":"widesearch-item-f1:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIFdpZGVTZWFyY2ggKGl0ZW0tZjEpLiBUaGlua2luZyBtb2RlOyB0ZW1wZXJhdHVyZSAxLjAsIHRvcC1wIDEuMCwgY29udGV4dCAyNjIxNDQuIENvbXBhcmF0b3Igc2V0dGluZ3M6IEdQVC01LjQgeGhpZ2gsIE9wdXM0LjYgbWF4LCBHZW1pbmkzLjFQcm8gaGlnaC4gU2VlIGJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMuIE9ubHkgb3duIEsyLjYgYW5kIHN0YXJyZWQgcmUtZXZhbHVhdGlvbnMgY2FuIGZvcm0gbmV3IGNvbXBhcmlzb25zLg","id":"launch-kimi-k26-card-2126","ingestRunId":"launch-kimi-k26-card","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"agentic","comparable":true,"configuration":"Kimi K2.6 card, WideSearch (item-f1). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons.","family":"WideSearch (item-f1)","locator":"Evaluation Results table, WideSearch (item-f1), column 2","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":80.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"toolathlon","evidenceKind":"lab_self_report","harnessId":"toolathlon:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIFRvb2xhdGhsb24uIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMu","id":"launch-kimi-k26-card-2127","ingestRunId":"launch-kimi-k26-card","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"agentic","comparable":true,"configuration":"Kimi K2.6 card, Toolathlon. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons.","family":"Toolathlon","locator":"Evaluation Results table, Toolathlon, column 2","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":50,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"toolathlon","evidenceKind":"lab_self_report","harnessId":"toolathlon:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIFRvb2xhdGhsb24uIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMu","id":"launch-kimi-k26-card-2128","ingestRunId":"launch-kimi-k26-card","modelId":"gpt-5.4","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"agentic","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, Toolathlon. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons.","family":"Toolathlon","locator":"Evaluation Results table, Toolathlon, column 3","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":54.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"toolathlon","evidenceKind":"lab_self_report","harnessId":"toolathlon:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIFRvb2xhdGhsb24uIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMu","id":"launch-kimi-k26-card-2129","ingestRunId":"launch-kimi-k26-card","modelId":"claude-opus-4-6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"agentic","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, Toolathlon. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons.","family":"Toolathlon","locator":"Evaluation Results table, Toolathlon, column 4","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":47.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"toolathlon","evidenceKind":"lab_self_report","harnessId":"toolathlon:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIFRvb2xhdGhsb24uIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMu","id":"launch-kimi-k26-card-2130","ingestRunId":"launch-kimi-k26-card","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"agentic","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, Toolathlon. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons.","family":"Toolathlon","locator":"Evaluation Results table, Toolathlon, column 5","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":48.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"mcpmark","evidenceKind":"lab_self_report","harnessId":"mcpmark:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIE1DUE1hcmsuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMu","id":"launch-kimi-k26-card-2131","ingestRunId":"launch-kimi-k26-card","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"agentic","comparable":true,"configuration":"Kimi K2.6 card, MCPMark. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons.","family":"MCPMark","locator":"Evaluation Results table, MCPMark, column 2","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":55.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"mcpmark","evidenceKind":"lab_self_report","harnessId":"mcpmark:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIE1DUE1hcmsuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMu","id":"launch-kimi-k26-card-2132","ingestRunId":"launch-kimi-k26-card","modelId":"gpt-5.4","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"agentic","comparable":true,"configuration":"Kimi K2.6 card, MCPMark. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons.","family":"MCPMark","locator":"Evaluation Results table, MCPMark, column 3","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":62.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"mcpmark","evidenceKind":"lab_self_report","harnessId":"mcpmark:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIE1DUE1hcmsuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMu","id":"launch-kimi-k26-card-2133","ingestRunId":"launch-kimi-k26-card","modelId":"claude-opus-4-6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"agentic","comparable":true,"configuration":"Kimi K2.6 card, MCPMark. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons.","family":"MCPMark","locator":"Evaluation Results table, MCPMark, column 4","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":56.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"mcpmark","evidenceKind":"lab_self_report","harnessId":"mcpmark:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIE1DUE1hcmsuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMu","id":"launch-kimi-k26-card-2134","ingestRunId":"launch-kimi-k26-card","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"agentic","comparable":true,"configuration":"Kimi K2.6 card, MCPMark. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons.","family":"MCPMark","locator":"Evaluation Results table, MCPMark, column 5","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":55.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"claw-eval-pass-3-percent","evidenceKind":"lab_self_report","harnessId":"claw-eval-pass-3-percent:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIENsYXcgRXZhbCAocGFzc14zKS4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4","id":"launch-kimi-k26-card-2135","ingestRunId":"launch-kimi-k26-card","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"agentic","comparable":true,"configuration":"Kimi K2.6 card, Claw Eval (pass^3). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons.","family":"Claw Eval (pass^3)","locator":"Evaluation Results table, Claw Eval (pass^3), column 2","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":62.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"claw-eval-pass-3-percent","evidenceKind":"lab_self_report","harnessId":"claw-eval-pass-3-percent:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIENsYXcgRXZhbCAocGFzc14zKS4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4","id":"launch-kimi-k26-card-2136","ingestRunId":"launch-kimi-k26-card","modelId":"gpt-5.4","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"agentic","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, Claw Eval (pass^3). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons.","family":"Claw Eval (pass^3)","locator":"Evaluation Results table, Claw Eval (pass^3), column 3","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":60.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"claw-eval-pass-3-percent","evidenceKind":"lab_self_report","harnessId":"claw-eval-pass-3-percent:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIENsYXcgRXZhbCAocGFzc14zKS4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4","id":"launch-kimi-k26-card-2137","ingestRunId":"launch-kimi-k26-card","modelId":"claude-opus-4-6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"agentic","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, Claw Eval (pass^3). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons.","family":"Claw Eval (pass^3)","locator":"Evaluation Results table, Claw Eval (pass^3), column 4","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":70.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"claw-eval-pass-3-percent","evidenceKind":"lab_self_report","harnessId":"claw-eval-pass-3-percent:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIENsYXcgRXZhbCAocGFzc14zKS4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4","id":"launch-kimi-k26-card-2138","ingestRunId":"launch-kimi-k26-card","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"agentic","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, Claw Eval (pass^3). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons.","family":"Claw Eval (pass^3)","locator":"Evaluation Results table, Claw Eval (pass^3), column 5","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":57.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"claw-eval-pass-3","evidenceKind":"lab_self_report","harnessId":"claw-eval-pass-3:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIENsYXcgRXZhbCAocGFzc0AzKS4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4","id":"launch-kimi-k26-card-2139","ingestRunId":"launch-kimi-k26-card","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"agentic","comparable":true,"configuration":"Kimi K2.6 card, Claw Eval (pass@3). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons.","family":"Claw Eval (pass@3)","locator":"Evaluation Results table, Claw Eval (pass@3), column 2","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":80.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"claw-eval-pass-3","evidenceKind":"lab_self_report","harnessId":"claw-eval-pass-3:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIENsYXcgRXZhbCAocGFzc0AzKS4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4","id":"launch-kimi-k26-card-2140","ingestRunId":"launch-kimi-k26-card","modelId":"gpt-5.4","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"agentic","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, Claw Eval (pass@3). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons.","family":"Claw Eval (pass@3)","locator":"Evaluation Results table, Claw Eval (pass@3), column 3","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":78.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"claw-eval-pass-3","evidenceKind":"lab_self_report","harnessId":"claw-eval-pass-3:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIENsYXcgRXZhbCAocGFzc0AzKS4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4","id":"launch-kimi-k26-card-2141","ingestRunId":"launch-kimi-k26-card","modelId":"claude-opus-4-6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"agentic","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, Claw Eval (pass@3). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons.","family":"Claw Eval (pass@3)","locator":"Evaluation Results table, Claw Eval (pass@3), column 4","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":82.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"claw-eval-pass-3","evidenceKind":"lab_self_report","harnessId":"claw-eval-pass-3:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIENsYXcgRXZhbCAocGFzc0AzKS4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4","id":"launch-kimi-k26-card-2142","ingestRunId":"launch-kimi-k26-card","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"agentic","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, Claw Eval (pass@3). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons.","family":"Claw Eval (pass@3)","locator":"Evaluation Results table, Claw Eval (pass@3), column 5","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":82.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"apex-agents","evidenceKind":"lab_self_report","harnessId":"apex-agents:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEFQRVgtQWdlbnRzLiBUaGlua2luZyBtb2RlOyB0ZW1wZXJhdHVyZSAxLjAsIHRvcC1wIDEuMCwgY29udGV4dCAyNjIxNDQuIENvbXBhcmF0b3Igc2V0dGluZ3M6IEdQVC01LjQgeGhpZ2gsIE9wdXM0LjYgbWF4LCBHZW1pbmkzLjFQcm8gaGlnaC4gU2VlIGJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMuIE9ubHkgb3duIEsyLjYgYW5kIHN0YXJyZWQgcmUtZXZhbHVhdGlvbnMgY2FuIGZvcm0gbmV3IGNvbXBhcmlzb25zLg","id":"launch-kimi-k26-card-2143","ingestRunId":"launch-kimi-k26-card","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"agentic","comparable":true,"configuration":"Kimi K2.6 card, APEX-Agents. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons.","family":"APEX-Agents","locator":"Evaluation Results table, APEX-Agents, column 2","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":27.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"apex-agents","evidenceKind":"lab_self_report","harnessId":"apex-agents:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEFQRVgtQWdlbnRzLiBUaGlua2luZyBtb2RlOyB0ZW1wZXJhdHVyZSAxLjAsIHRvcC1wIDEuMCwgY29udGV4dCAyNjIxNDQuIENvbXBhcmF0b3Igc2V0dGluZ3M6IEdQVC01LjQgeGhpZ2gsIE9wdXM0LjYgbWF4LCBHZW1pbmkzLjFQcm8gaGlnaC4gU2VlIGJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMuIE9ubHkgb3duIEsyLjYgYW5kIHN0YXJyZWQgcmUtZXZhbHVhdGlvbnMgY2FuIGZvcm0gbmV3IGNvbXBhcmlzb25zLg","id":"launch-kimi-k26-card-2144","ingestRunId":"launch-kimi-k26-card","modelId":"gpt-5.4","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"agentic","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, APEX-Agents. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons.","family":"APEX-Agents","locator":"Evaluation Results table, APEX-Agents, column 3","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":33.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"apex-agents","evidenceKind":"lab_self_report","harnessId":"apex-agents:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEFQRVgtQWdlbnRzLiBUaGlua2luZyBtb2RlOyB0ZW1wZXJhdHVyZSAxLjAsIHRvcC1wIDEuMCwgY29udGV4dCAyNjIxNDQuIENvbXBhcmF0b3Igc2V0dGluZ3M6IEdQVC01LjQgeGhpZ2gsIE9wdXM0LjYgbWF4LCBHZW1pbmkzLjFQcm8gaGlnaC4gU2VlIGJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMuIE9ubHkgb3duIEsyLjYgYW5kIHN0YXJyZWQgcmUtZXZhbHVhdGlvbnMgY2FuIGZvcm0gbmV3IGNvbXBhcmlzb25zLg","id":"launch-kimi-k26-card-2145","ingestRunId":"launch-kimi-k26-card","modelId":"claude-opus-4-6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"agentic","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, APEX-Agents. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons.","family":"APEX-Agents","locator":"Evaluation Results table, APEX-Agents, column 4","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":33,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"apex-agents","evidenceKind":"lab_self_report","harnessId":"apex-agents:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEFQRVgtQWdlbnRzLiBUaGlua2luZyBtb2RlOyB0ZW1wZXJhdHVyZSAxLjAsIHRvcC1wIDEuMCwgY29udGV4dCAyNjIxNDQuIENvbXBhcmF0b3Igc2V0dGluZ3M6IEdQVC01LjQgeGhpZ2gsIE9wdXM0LjYgbWF4LCBHZW1pbmkzLjFQcm8gaGlnaC4gU2VlIGJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMuIE9ubHkgb3duIEsyLjYgYW5kIHN0YXJyZWQgcmUtZXZhbHVhdGlvbnMgY2FuIGZvcm0gbmV3IGNvbXBhcmlzb25zLg","id":"launch-kimi-k26-card-2146","ingestRunId":"launch-kimi-k26-card","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"agentic","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, APEX-Agents. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons.","family":"APEX-Agents","locator":"Evaluation Results table, APEX-Agents, column 5","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":32,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"osworld-verified-percent","evidenceKind":"lab_self_report","harnessId":"osworld-verified-percent:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIE9TV29ybGQtVmVyaWZpZWQuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMu","id":"launch-kimi-k26-card-2147","ingestRunId":"launch-kimi-k26-card","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"agentic","comparable":true,"configuration":"Kimi K2.6 card, OSWorld-Verified. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons.","family":"OSWorld","locator":"Evaluation Results table, OSWorld-Verified, column 2","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":73.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"osworld-verified-percent","evidenceKind":"lab_self_report","harnessId":"osworld-verified-percent:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIE9TV29ybGQtVmVyaWZpZWQuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMu","id":"launch-kimi-k26-card-2148","ingestRunId":"launch-kimi-k26-card","modelId":"gpt-5.4","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"agentic","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, OSWorld-Verified. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons.","family":"OSWorld","locator":"Evaluation Results table, OSWorld-Verified, column 3","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":75,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"osworld-verified-percent","evidenceKind":"lab_self_report","harnessId":"osworld-verified-percent:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIE9TV29ybGQtVmVyaWZpZWQuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMu","id":"launch-kimi-k26-card-2149","ingestRunId":"launch-kimi-k26-card","modelId":"claude-opus-4-6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"agentic","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, OSWorld-Verified. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons.","family":"OSWorld","locator":"Evaluation Results table, OSWorld-Verified, column 4","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":72.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"terminal-bench-2-0-terminus-2","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-0-terminus-2:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIFRlcm1pbmFsLUJlbmNoIDIuMCAoVGVybWludXMtMikuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMuIENvZGluZzogYXZlcmFnZSBvZiB0ZW4gcnVuczsgVGVybWluYWwgdXNlcyBUZXJtaW51czIgcHJlc2VydmUtdGhpbmtpbmc7IFNXRSB1c2VzIGluLWhvdXNlIFNXRS1hZ2VudC4","id":"launch-kimi-k26-card-2150","ingestRunId":"launch-kimi-k26-card","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"coding","comparable":true,"configuration":"Kimi K2.6 card, Terminal-Bench 2.0 (Terminus-2). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent.","family":"Terminal-Bench","locator":"Evaluation Results table, Terminal-Bench 2.0 (Terminus-2), column 2","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":66.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"terminal-bench-2-0-terminus-2","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-0-terminus-2:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIFRlcm1pbmFsLUJlbmNoIDIuMCAoVGVybWludXMtMikuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMuIENvZGluZzogYXZlcmFnZSBvZiB0ZW4gcnVuczsgVGVybWluYWwgdXNlcyBUZXJtaW51czIgcHJlc2VydmUtdGhpbmtpbmc7IFNXRSB1c2VzIGluLWhvdXNlIFNXRS1hZ2VudC4","id":"launch-kimi-k26-card-2151","ingestRunId":"launch-kimi-k26-card","modelId":"gpt-5.4","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"coding","comparable":true,"configuration":"Kimi K2.6 card, Terminal-Bench 2.0 (Terminus-2). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent.","family":"Terminal-Bench","locator":"Evaluation Results table, Terminal-Bench 2.0 (Terminus-2), column 3","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":65.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"terminal-bench-2-0-terminus-2","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-0-terminus-2:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIFRlcm1pbmFsLUJlbmNoIDIuMCAoVGVybWludXMtMikuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMuIENvZGluZzogYXZlcmFnZSBvZiB0ZW4gcnVuczsgVGVybWluYWwgdXNlcyBUZXJtaW51czIgcHJlc2VydmUtdGhpbmtpbmc7IFNXRSB1c2VzIGluLWhvdXNlIFNXRS1hZ2VudC4","id":"launch-kimi-k26-card-2152","ingestRunId":"launch-kimi-k26-card","modelId":"claude-opus-4-6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"coding","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, Terminal-Bench 2.0 (Terminus-2). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent.","family":"Terminal-Bench","locator":"Evaluation Results table, Terminal-Bench 2.0 (Terminus-2), column 4","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":65.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"terminal-bench-2-0-terminus-2","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-0-terminus-2:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIFRlcm1pbmFsLUJlbmNoIDIuMCAoVGVybWludXMtMikuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMuIENvZGluZzogYXZlcmFnZSBvZiB0ZW4gcnVuczsgVGVybWluYWwgdXNlcyBUZXJtaW51czIgcHJlc2VydmUtdGhpbmtpbmc7IFNXRSB1c2VzIGluLWhvdXNlIFNXRS1hZ2VudC4","id":"launch-kimi-k26-card-2153","ingestRunId":"launch-kimi-k26-card","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"coding","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, Terminal-Bench 2.0 (Terminus-2). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent.","family":"Terminal-Bench","locator":"Evaluation Results table, Terminal-Bench 2.0 (Terminus-2), column 5","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":68.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"swe-bench-pro-percent","evidenceKind":"lab_self_report","harnessId":"swe-bench-pro-percent:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIFNXRS1CZW5jaCBQcm8uIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMuIENvZGluZzogYXZlcmFnZSBvZiB0ZW4gcnVuczsgVGVybWluYWwgdXNlcyBUZXJtaW51czIgcHJlc2VydmUtdGhpbmtpbmc7IFNXRSB1c2VzIGluLWhvdXNlIFNXRS1hZ2VudC4","id":"launch-kimi-k26-card-2154","ingestRunId":"launch-kimi-k26-card","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"coding","comparable":true,"configuration":"Kimi K2.6 card, SWE-Bench Pro. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent.","family":"SWE-bench Pro","locator":"Evaluation Results table, SWE-Bench Pro, column 2","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":58.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"swe-bench-pro-percent","evidenceKind":"lab_self_report","harnessId":"swe-bench-pro-percent:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIFNXRS1CZW5jaCBQcm8uIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMuIENvZGluZzogYXZlcmFnZSBvZiB0ZW4gcnVuczsgVGVybWluYWwgdXNlcyBUZXJtaW51czIgcHJlc2VydmUtdGhpbmtpbmc7IFNXRSB1c2VzIGluLWhvdXNlIFNXRS1hZ2VudC4","id":"launch-kimi-k26-card-2155","ingestRunId":"launch-kimi-k26-card","modelId":"gpt-5.4","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"coding","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, SWE-Bench Pro. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent.","family":"SWE-bench Pro","locator":"Evaluation Results table, SWE-Bench Pro, column 3","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":57.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"swe-bench-pro-percent","evidenceKind":"lab_self_report","harnessId":"swe-bench-pro-percent:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIFNXRS1CZW5jaCBQcm8uIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMuIENvZGluZzogYXZlcmFnZSBvZiB0ZW4gcnVuczsgVGVybWluYWwgdXNlcyBUZXJtaW51czIgcHJlc2VydmUtdGhpbmtpbmc7IFNXRSB1c2VzIGluLWhvdXNlIFNXRS1hZ2VudC4","id":"launch-kimi-k26-card-2156","ingestRunId":"launch-kimi-k26-card","modelId":"claude-opus-4-6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"coding","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, SWE-Bench Pro. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent.","family":"SWE-bench Pro","locator":"Evaluation Results table, SWE-Bench Pro, column 4","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":53.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"swe-bench-pro-percent","evidenceKind":"lab_self_report","harnessId":"swe-bench-pro-percent:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIFNXRS1CZW5jaCBQcm8uIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMuIENvZGluZzogYXZlcmFnZSBvZiB0ZW4gcnVuczsgVGVybWluYWwgdXNlcyBUZXJtaW51czIgcHJlc2VydmUtdGhpbmtpbmc7IFNXRSB1c2VzIGluLWhvdXNlIFNXRS1hZ2VudC4","id":"launch-kimi-k26-card-2157","ingestRunId":"launch-kimi-k26-card","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"coding","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, SWE-Bench Pro. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent.","family":"SWE-bench Pro","locator":"Evaluation Results table, SWE-Bench Pro, column 5","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":54.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"swe-bench-multilingual","evidenceKind":"lab_self_report","harnessId":"swe-bench-multilingual:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIFNXRS1CZW5jaCBNdWx0aWxpbmd1YWwuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMuIENvZGluZzogYXZlcmFnZSBvZiB0ZW4gcnVuczsgVGVybWluYWwgdXNlcyBUZXJtaW51czIgcHJlc2VydmUtdGhpbmtpbmc7IFNXRSB1c2VzIGluLWhvdXNlIFNXRS1hZ2VudC4","id":"launch-kimi-k26-card-2158","ingestRunId":"launch-kimi-k26-card","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"coding","comparable":true,"configuration":"Kimi K2.6 card, SWE-Bench Multilingual. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent.","family":"SWE-bench Multilingual","locator":"Evaluation Results table, SWE-Bench Multilingual, column 2","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":76.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"swe-bench-multilingual","evidenceKind":"lab_self_report","harnessId":"swe-bench-multilingual:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIFNXRS1CZW5jaCBNdWx0aWxpbmd1YWwuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMuIENvZGluZzogYXZlcmFnZSBvZiB0ZW4gcnVuczsgVGVybWluYWwgdXNlcyBUZXJtaW51czIgcHJlc2VydmUtdGhpbmtpbmc7IFNXRSB1c2VzIGluLWhvdXNlIFNXRS1hZ2VudC4","id":"launch-kimi-k26-card-2159","ingestRunId":"launch-kimi-k26-card","modelId":"claude-opus-4-6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"coding","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, SWE-Bench Multilingual. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent.","family":"SWE-bench Multilingual","locator":"Evaluation Results table, SWE-Bench Multilingual, column 4","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":77.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"swe-bench-multilingual","evidenceKind":"lab_self_report","harnessId":"swe-bench-multilingual:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIFNXRS1CZW5jaCBNdWx0aWxpbmd1YWwuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMuIENvZGluZzogYXZlcmFnZSBvZiB0ZW4gcnVuczsgVGVybWluYWwgdXNlcyBUZXJtaW51czIgcHJlc2VydmUtdGhpbmtpbmc7IFNXRSB1c2VzIGluLWhvdXNlIFNXRS1hZ2VudC4","id":"launch-kimi-k26-card-2160","ingestRunId":"launch-kimi-k26-card","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"coding","comparable":true,"configuration":"Kimi K2.6 card, SWE-Bench Multilingual. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent.","family":"SWE-bench Multilingual","locator":"Evaluation Results table, SWE-Bench Multilingual, column 5","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":76.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"swe-bench-verified-percent","evidenceKind":"lab_self_report","harnessId":"swe-bench-verified-percent:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIFNXRS1CZW5jaCBWZXJpZmllZC4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4gQ29kaW5nOiBhdmVyYWdlIG9mIHRlbiBydW5zOyBUZXJtaW5hbCB1c2VzIFRlcm1pbnVzMiBwcmVzZXJ2ZS10aGlua2luZzsgU1dFIHVzZXMgaW4taG91c2UgU1dFLWFnZW50Lg","id":"launch-kimi-k26-card-2161","ingestRunId":"launch-kimi-k26-card","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"coding","comparable":true,"configuration":"Kimi K2.6 card, SWE-Bench Verified. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent.","family":"SWE-bench Verified","locator":"Evaluation Results table, SWE-Bench Verified, column 2","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":80.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"swe-bench-verified-percent","evidenceKind":"lab_self_report","harnessId":"swe-bench-verified-percent:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIFNXRS1CZW5jaCBWZXJpZmllZC4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4gQ29kaW5nOiBhdmVyYWdlIG9mIHRlbiBydW5zOyBUZXJtaW5hbCB1c2VzIFRlcm1pbnVzMiBwcmVzZXJ2ZS10aGlua2luZzsgU1dFIHVzZXMgaW4taG91c2UgU1dFLWFnZW50Lg","id":"launch-kimi-k26-card-2162","ingestRunId":"launch-kimi-k26-card","modelId":"claude-opus-4-6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"coding","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, SWE-Bench Verified. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent.","family":"SWE-bench Verified","locator":"Evaluation Results table, SWE-Bench Verified, column 4","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":80.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"swe-bench-verified-percent","evidenceKind":"lab_self_report","harnessId":"swe-bench-verified-percent:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIFNXRS1CZW5jaCBWZXJpZmllZC4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4gQ29kaW5nOiBhdmVyYWdlIG9mIHRlbiBydW5zOyBUZXJtaW5hbCB1c2VzIFRlcm1pbnVzMiBwcmVzZXJ2ZS10aGlua2luZzsgU1dFIHVzZXMgaW4taG91c2UgU1dFLWFnZW50Lg","id":"launch-kimi-k26-card-2163","ingestRunId":"launch-kimi-k26-card","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"coding","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, SWE-Bench Verified. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent.","family":"SWE-bench Verified","locator":"Evaluation Results table, SWE-Bench Verified, column 5","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":80.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"scicode","evidenceKind":"lab_self_report","harnessId":"scicode:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIFNjaUNvZGUuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMuIENvZGluZzogYXZlcmFnZSBvZiB0ZW4gcnVuczsgVGVybWluYWwgdXNlcyBUZXJtaW51czIgcHJlc2VydmUtdGhpbmtpbmc7IFNXRSB1c2VzIGluLWhvdXNlIFNXRS1hZ2VudC4","id":"launch-kimi-k26-card-2164","ingestRunId":"launch-kimi-k26-card","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"coding","comparable":true,"configuration":"Kimi K2.6 card, SciCode. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent.","family":"SciCode","locator":"Evaluation Results table, SciCode, column 2","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":52.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"scicode","evidenceKind":"lab_self_report","harnessId":"scicode:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIFNjaUNvZGUuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMuIENvZGluZzogYXZlcmFnZSBvZiB0ZW4gcnVuczsgVGVybWluYWwgdXNlcyBUZXJtaW51czIgcHJlc2VydmUtdGhpbmtpbmc7IFNXRSB1c2VzIGluLWhvdXNlIFNXRS1hZ2VudC4","id":"launch-kimi-k26-card-2165","ingestRunId":"launch-kimi-k26-card","modelId":"gpt-5.4","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"coding","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, SciCode. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent.","family":"SciCode","locator":"Evaluation Results table, SciCode, column 3","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":56.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"scicode","evidenceKind":"lab_self_report","harnessId":"scicode:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIFNjaUNvZGUuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMuIENvZGluZzogYXZlcmFnZSBvZiB0ZW4gcnVuczsgVGVybWluYWwgdXNlcyBUZXJtaW51czIgcHJlc2VydmUtdGhpbmtpbmc7IFNXRSB1c2VzIGluLWhvdXNlIFNXRS1hZ2VudC4","id":"launch-kimi-k26-card-2166","ingestRunId":"launch-kimi-k26-card","modelId":"claude-opus-4-6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"coding","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, SciCode. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent.","family":"SciCode","locator":"Evaluation Results table, SciCode, column 4","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":51.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"scicode","evidenceKind":"lab_self_report","harnessId":"scicode:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIFNjaUNvZGUuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMuIENvZGluZzogYXZlcmFnZSBvZiB0ZW4gcnVuczsgVGVybWluYWwgdXNlcyBUZXJtaW51czIgcHJlc2VydmUtdGhpbmtpbmc7IFNXRSB1c2VzIGluLWhvdXNlIFNXRS1hZ2VudC4","id":"launch-kimi-k26-card-2167","ingestRunId":"launch-kimi-k26-card","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"coding","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, SciCode. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent.","family":"SciCode","locator":"Evaluation Results table, SciCode, column 5","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":58.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"ojbench-python","evidenceKind":"lab_self_report","harnessId":"ojbench-python:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIE9KQmVuY2ggKHB5dGhvbikuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMuIENvZGluZzogYXZlcmFnZSBvZiB0ZW4gcnVuczsgVGVybWluYWwgdXNlcyBUZXJtaW51czIgcHJlc2VydmUtdGhpbmtpbmc7IFNXRSB1c2VzIGluLWhvdXNlIFNXRS1hZ2VudC4","id":"launch-kimi-k26-card-2168","ingestRunId":"launch-kimi-k26-card","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"coding","comparable":true,"configuration":"Kimi K2.6 card, OJBench (python). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent.","family":"OJBench (python)","locator":"Evaluation Results table, OJBench (python), column 2","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":60.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"ojbench-python","evidenceKind":"lab_self_report","harnessId":"ojbench-python:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIE9KQmVuY2ggKHB5dGhvbikuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMuIENvZGluZzogYXZlcmFnZSBvZiB0ZW4gcnVuczsgVGVybWluYWwgdXNlcyBUZXJtaW51czIgcHJlc2VydmUtdGhpbmtpbmc7IFNXRSB1c2VzIGluLWhvdXNlIFNXRS1hZ2VudC4","id":"launch-kimi-k26-card-2169","ingestRunId":"launch-kimi-k26-card","modelId":"claude-opus-4-6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"coding","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, OJBench (python). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent.","family":"OJBench (python)","locator":"Evaluation Results table, OJBench (python), column 4","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":60.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"ojbench-python","evidenceKind":"lab_self_report","harnessId":"ojbench-python:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIE9KQmVuY2ggKHB5dGhvbikuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMuIENvZGluZzogYXZlcmFnZSBvZiB0ZW4gcnVuczsgVGVybWluYWwgdXNlcyBUZXJtaW51czIgcHJlc2VydmUtdGhpbmtpbmc7IFNXRSB1c2VzIGluLWhvdXNlIFNXRS1hZ2VudC4","id":"launch-kimi-k26-card-2170","ingestRunId":"launch-kimi-k26-card","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"coding","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, OJBench (python). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent.","family":"OJBench (python)","locator":"Evaluation Results table, OJBench (python), column 5","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":70.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"livecodebench-v6","evidenceKind":"lab_self_report","harnessId":"livecodebench-v6:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIExpdmVDb2RlQmVuY2ggKHY2KS4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4gQ29kaW5nOiBhdmVyYWdlIG9mIHRlbiBydW5zOyBUZXJtaW5hbCB1c2VzIFRlcm1pbnVzMiBwcmVzZXJ2ZS10aGlua2luZzsgU1dFIHVzZXMgaW4taG91c2UgU1dFLWFnZW50Lg","id":"launch-kimi-k26-card-2171","ingestRunId":"launch-kimi-k26-card","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"coding","comparable":true,"configuration":"Kimi K2.6 card, LiveCodeBench (v6). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent.","family":"LiveCodeBench","locator":"Evaluation Results table, LiveCodeBench (v6), column 2","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":89.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"livecodebench-v6","evidenceKind":"lab_self_report","harnessId":"livecodebench-v6:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIExpdmVDb2RlQmVuY2ggKHY2KS4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4gQ29kaW5nOiBhdmVyYWdlIG9mIHRlbiBydW5zOyBUZXJtaW5hbCB1c2VzIFRlcm1pbnVzMiBwcmVzZXJ2ZS10aGlua2luZzsgU1dFIHVzZXMgaW4taG91c2UgU1dFLWFnZW50Lg","id":"launch-kimi-k26-card-2172","ingestRunId":"launch-kimi-k26-card","modelId":"claude-opus-4-6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"coding","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, LiveCodeBench (v6). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent.","family":"LiveCodeBench","locator":"Evaluation Results table, LiveCodeBench (v6), column 4","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":88.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"livecodebench-v6","evidenceKind":"lab_self_report","harnessId":"livecodebench-v6:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIExpdmVDb2RlQmVuY2ggKHY2KS4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4gQ29kaW5nOiBhdmVyYWdlIG9mIHRlbiBydW5zOyBUZXJtaW5hbCB1c2VzIFRlcm1pbnVzMiBwcmVzZXJ2ZS10aGlua2luZzsgU1dFIHVzZXMgaW4taG91c2UgU1dFLWFnZW50Lg","id":"launch-kimi-k26-card-2173","ingestRunId":"launch-kimi-k26-card","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"coding","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, LiveCodeBench (v6). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent.","family":"LiveCodeBench","locator":"Evaluation Results table, LiveCodeBench (v6), column 5","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":91.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"hle-full","evidenceKind":"lab_self_report","harnessId":"hle-full:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEhMRS1GdWxsLiBUaGlua2luZyBtb2RlOyB0ZW1wZXJhdHVyZSAxLjAsIHRvcC1wIDEuMCwgY29udGV4dCAyNjIxNDQuIENvbXBhcmF0b3Igc2V0dGluZ3M6IEdQVC01LjQgeGhpZ2gsIE9wdXM0LjYgbWF4LCBHZW1pbmkzLjFQcm8gaGlnaC4gU2VlIGJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMuIE9ubHkgb3duIEsyLjYgYW5kIHN0YXJyZWQgcmUtZXZhbHVhdGlvbnMgY2FuIGZvcm0gbmV3IGNvbXBhcmlzb25zLiBSZWFzb25pbmc6IDk4MzA0IGdlbmVyYXRpb24gdG9rZW5zOyBITEUgZnVsbCBzZXQu","id":"launch-kimi-k26-card-2174","ingestRunId":"launch-kimi-k26-card","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"hard_reasoning","comparable":true,"configuration":"Kimi K2.6 card, HLE-Full. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Reasoning: 98304 generation tokens; HLE full set.","family":"Humanity's Last Exam","locator":"Evaluation Results table, HLE-Full, column 2","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":34.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"hle-full","evidenceKind":"lab_self_report","harnessId":"hle-full:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEhMRS1GdWxsLiBUaGlua2luZyBtb2RlOyB0ZW1wZXJhdHVyZSAxLjAsIHRvcC1wIDEuMCwgY29udGV4dCAyNjIxNDQuIENvbXBhcmF0b3Igc2V0dGluZ3M6IEdQVC01LjQgeGhpZ2gsIE9wdXM0LjYgbWF4LCBHZW1pbmkzLjFQcm8gaGlnaC4gU2VlIGJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMuIE9ubHkgb3duIEsyLjYgYW5kIHN0YXJyZWQgcmUtZXZhbHVhdGlvbnMgY2FuIGZvcm0gbmV3IGNvbXBhcmlzb25zLiBSZWFzb25pbmc6IDk4MzA0IGdlbmVyYXRpb24gdG9rZW5zOyBITEUgZnVsbCBzZXQu","id":"launch-kimi-k26-card-2175","ingestRunId":"launch-kimi-k26-card","modelId":"gpt-5.4","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"hard_reasoning","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, HLE-Full. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Reasoning: 98304 generation tokens; HLE full set.","family":"Humanity's Last Exam","locator":"Evaluation Results table, HLE-Full, column 3","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":39.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"hle-full","evidenceKind":"lab_self_report","harnessId":"hle-full:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEhMRS1GdWxsLiBUaGlua2luZyBtb2RlOyB0ZW1wZXJhdHVyZSAxLjAsIHRvcC1wIDEuMCwgY29udGV4dCAyNjIxNDQuIENvbXBhcmF0b3Igc2V0dGluZ3M6IEdQVC01LjQgeGhpZ2gsIE9wdXM0LjYgbWF4LCBHZW1pbmkzLjFQcm8gaGlnaC4gU2VlIGJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMuIE9ubHkgb3duIEsyLjYgYW5kIHN0YXJyZWQgcmUtZXZhbHVhdGlvbnMgY2FuIGZvcm0gbmV3IGNvbXBhcmlzb25zLiBSZWFzb25pbmc6IDk4MzA0IGdlbmVyYXRpb24gdG9rZW5zOyBITEUgZnVsbCBzZXQu","id":"launch-kimi-k26-card-2176","ingestRunId":"launch-kimi-k26-card","modelId":"claude-opus-4-6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"hard_reasoning","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, HLE-Full. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Reasoning: 98304 generation tokens; HLE full set.","family":"Humanity's Last Exam","locator":"Evaluation Results table, HLE-Full, column 4","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":40,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"hle-full","evidenceKind":"lab_self_report","harnessId":"hle-full:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEhMRS1GdWxsLiBUaGlua2luZyBtb2RlOyB0ZW1wZXJhdHVyZSAxLjAsIHRvcC1wIDEuMCwgY29udGV4dCAyNjIxNDQuIENvbXBhcmF0b3Igc2V0dGluZ3M6IEdQVC01LjQgeGhpZ2gsIE9wdXM0LjYgbWF4LCBHZW1pbmkzLjFQcm8gaGlnaC4gU2VlIGJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMuIE9ubHkgb3duIEsyLjYgYW5kIHN0YXJyZWQgcmUtZXZhbHVhdGlvbnMgY2FuIGZvcm0gbmV3IGNvbXBhcmlzb25zLiBSZWFzb25pbmc6IDk4MzA0IGdlbmVyYXRpb24gdG9rZW5zOyBITEUgZnVsbCBzZXQu","id":"launch-kimi-k26-card-2177","ingestRunId":"launch-kimi-k26-card","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"hard_reasoning","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, HLE-Full. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Reasoning: 98304 generation tokens; HLE full set.","family":"Humanity's Last Exam","locator":"Evaluation Results table, HLE-Full, column 5","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":44.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"aime-2026","evidenceKind":"lab_self_report","harnessId":"aime-2026:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEFJTUUgMjAyNi4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4gUmVhc29uaW5nOiA5ODMwNCBnZW5lcmF0aW9uIHRva2VuczsgSExFIGZ1bGwgc2V0Lg","id":"launch-kimi-k26-card-2178","ingestRunId":"launch-kimi-k26-card","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"hard_reasoning","comparable":true,"configuration":"Kimi K2.6 card, AIME 2026. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Reasoning: 98304 generation tokens; HLE full set.","family":"AIME","locator":"Evaluation Results table, AIME 2026, column 2","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":96.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"aime-2026","evidenceKind":"lab_self_report","harnessId":"aime-2026:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEFJTUUgMjAyNi4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4gUmVhc29uaW5nOiA5ODMwNCBnZW5lcmF0aW9uIHRva2VuczsgSExFIGZ1bGwgc2V0Lg","id":"launch-kimi-k26-card-2179","ingestRunId":"launch-kimi-k26-card","modelId":"gpt-5.4","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"hard_reasoning","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, AIME 2026. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Reasoning: 98304 generation tokens; HLE full set.","family":"AIME","locator":"Evaluation Results table, AIME 2026, column 3","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":99.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"aime-2026","evidenceKind":"lab_self_report","harnessId":"aime-2026:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEFJTUUgMjAyNi4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4gUmVhc29uaW5nOiA5ODMwNCBnZW5lcmF0aW9uIHRva2VuczsgSExFIGZ1bGwgc2V0Lg","id":"launch-kimi-k26-card-2180","ingestRunId":"launch-kimi-k26-card","modelId":"claude-opus-4-6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"hard_reasoning","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, AIME 2026. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Reasoning: 98304 generation tokens; HLE full set.","family":"AIME","locator":"Evaluation Results table, AIME 2026, column 4","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":96.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"aime-2026","evidenceKind":"lab_self_report","harnessId":"aime-2026:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEFJTUUgMjAyNi4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4gUmVhc29uaW5nOiA5ODMwNCBnZW5lcmF0aW9uIHRva2VuczsgSExFIGZ1bGwgc2V0Lg","id":"launch-kimi-k26-card-2181","ingestRunId":"launch-kimi-k26-card","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"hard_reasoning","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, AIME 2026. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Reasoning: 98304 generation tokens; HLE full set.","family":"AIME","locator":"Evaluation Results table, AIME 2026, column 5","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":98.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"hmmt-2026-feb","evidenceKind":"lab_self_report","harnessId":"hmmt-2026-feb:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEhNTVQgMjAyNiAoRmViKS4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4gUmVhc29uaW5nOiA5ODMwNCBnZW5lcmF0aW9uIHRva2VuczsgSExFIGZ1bGwgc2V0Lg","id":"launch-kimi-k26-card-2182","ingestRunId":"launch-kimi-k26-card","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"hard_reasoning","comparable":true,"configuration":"Kimi K2.6 card, HMMT 2026 (Feb). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Reasoning: 98304 generation tokens; HLE full set.","family":"HMMT 2026 (Feb)","locator":"Evaluation Results table, HMMT 2026 (Feb), column 2","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":92.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"hmmt-2026-feb","evidenceKind":"lab_self_report","harnessId":"hmmt-2026-feb:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEhNTVQgMjAyNiAoRmViKS4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4gUmVhc29uaW5nOiA5ODMwNCBnZW5lcmF0aW9uIHRva2VuczsgSExFIGZ1bGwgc2V0Lg","id":"launch-kimi-k26-card-2183","ingestRunId":"launch-kimi-k26-card","modelId":"gpt-5.4","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"hard_reasoning","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, HMMT 2026 (Feb). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Reasoning: 98304 generation tokens; HLE full set.","family":"HMMT 2026 (Feb)","locator":"Evaluation Results table, HMMT 2026 (Feb), column 3","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":97.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"hmmt-2026-feb","evidenceKind":"lab_self_report","harnessId":"hmmt-2026-feb:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEhNTVQgMjAyNiAoRmViKS4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4gUmVhc29uaW5nOiA5ODMwNCBnZW5lcmF0aW9uIHRva2VuczsgSExFIGZ1bGwgc2V0Lg","id":"launch-kimi-k26-card-2184","ingestRunId":"launch-kimi-k26-card","modelId":"claude-opus-4-6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"hard_reasoning","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, HMMT 2026 (Feb). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Reasoning: 98304 generation tokens; HLE full set.","family":"HMMT 2026 (Feb)","locator":"Evaluation Results table, HMMT 2026 (Feb), column 4","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":96.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"hmmt-2026-feb","evidenceKind":"lab_self_report","harnessId":"hmmt-2026-feb:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEhNTVQgMjAyNiAoRmViKS4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4gUmVhc29uaW5nOiA5ODMwNCBnZW5lcmF0aW9uIHRva2VuczsgSExFIGZ1bGwgc2V0Lg","id":"launch-kimi-k26-card-2185","ingestRunId":"launch-kimi-k26-card","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"hard_reasoning","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, HMMT 2026 (Feb). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Reasoning: 98304 generation tokens; HLE full set.","family":"HMMT 2026 (Feb)","locator":"Evaluation Results table, HMMT 2026 (Feb), column 5","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":94.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"imo-answerbench","evidenceKind":"lab_self_report","harnessId":"imo-answerbench:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIElNTy1BbnN3ZXJCZW5jaC4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4gUmVhc29uaW5nOiA5ODMwNCBnZW5lcmF0aW9uIHRva2VuczsgSExFIGZ1bGwgc2V0Lg","id":"launch-kimi-k26-card-2186","ingestRunId":"launch-kimi-k26-card","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"hard_reasoning","comparable":true,"configuration":"Kimi K2.6 card, IMO-AnswerBench. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Reasoning: 98304 generation tokens; HLE full set.","family":"IMO-AnswerBench","locator":"Evaluation Results table, IMO-AnswerBench, column 2","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":86,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"imo-answerbench","evidenceKind":"lab_self_report","harnessId":"imo-answerbench:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIElNTy1BbnN3ZXJCZW5jaC4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4gUmVhc29uaW5nOiA5ODMwNCBnZW5lcmF0aW9uIHRva2VuczsgSExFIGZ1bGwgc2V0Lg","id":"launch-kimi-k26-card-2187","ingestRunId":"launch-kimi-k26-card","modelId":"gpt-5.4","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"hard_reasoning","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, IMO-AnswerBench. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Reasoning: 98304 generation tokens; HLE full set.","family":"IMO-AnswerBench","locator":"Evaluation Results table, IMO-AnswerBench, column 3","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":91.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"imo-answerbench","evidenceKind":"lab_self_report","harnessId":"imo-answerbench:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIElNTy1BbnN3ZXJCZW5jaC4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4gUmVhc29uaW5nOiA5ODMwNCBnZW5lcmF0aW9uIHRva2VuczsgSExFIGZ1bGwgc2V0Lg","id":"launch-kimi-k26-card-2188","ingestRunId":"launch-kimi-k26-card","modelId":"claude-opus-4-6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"hard_reasoning","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, IMO-AnswerBench. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Reasoning: 98304 generation tokens; HLE full set.","family":"IMO-AnswerBench","locator":"Evaluation Results table, IMO-AnswerBench, column 4","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":75.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"imo-answerbench","evidenceKind":"lab_self_report","harnessId":"imo-answerbench:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIElNTy1BbnN3ZXJCZW5jaC4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4gUmVhc29uaW5nOiA5ODMwNCBnZW5lcmF0aW9uIHRva2VuczsgSExFIGZ1bGwgc2V0Lg","id":"launch-kimi-k26-card-2189","ingestRunId":"launch-kimi-k26-card","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"hard_reasoning","comparable":true,"configuration":"Kimi K2.6 card, IMO-AnswerBench. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Reasoning: 98304 generation tokens; HLE full set.","family":"IMO-AnswerBench","locator":"Evaluation Results table, IMO-AnswerBench, column 5","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":91,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"gpqa-diamond-percent","evidenceKind":"lab_self_report","harnessId":"gpqa-diamond-percent:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEdQUUEtRGlhbW9uZC4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4gUmVhc29uaW5nOiA5ODMwNCBnZW5lcmF0aW9uIHRva2VuczsgSExFIGZ1bGwgc2V0Lg","id":"launch-kimi-k26-card-2190","ingestRunId":"launch-kimi-k26-card","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"hard_reasoning","comparable":true,"configuration":"Kimi K2.6 card, GPQA-Diamond. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Reasoning: 98304 generation tokens; HLE full set.","family":"GPQA-Diamond","locator":"Evaluation Results table, GPQA-Diamond, column 2","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":90.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"gpqa-diamond-percent","evidenceKind":"lab_self_report","harnessId":"gpqa-diamond-percent:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEdQUUEtRGlhbW9uZC4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4gUmVhc29uaW5nOiA5ODMwNCBnZW5lcmF0aW9uIHRva2VuczsgSExFIGZ1bGwgc2V0Lg","id":"launch-kimi-k26-card-2191","ingestRunId":"launch-kimi-k26-card","modelId":"gpt-5.4","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"hard_reasoning","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, GPQA-Diamond. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Reasoning: 98304 generation tokens; HLE full set.","family":"GPQA-Diamond","locator":"Evaluation Results table, GPQA-Diamond, column 3","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":92.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"gpqa-diamond-percent","evidenceKind":"lab_self_report","harnessId":"gpqa-diamond-percent:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEdQUUEtRGlhbW9uZC4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4gUmVhc29uaW5nOiA5ODMwNCBnZW5lcmF0aW9uIHRva2VuczsgSExFIGZ1bGwgc2V0Lg","id":"launch-kimi-k26-card-2192","ingestRunId":"launch-kimi-k26-card","modelId":"claude-opus-4-6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"hard_reasoning","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, GPQA-Diamond. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Reasoning: 98304 generation tokens; HLE full set.","family":"GPQA-Diamond","locator":"Evaluation Results table, GPQA-Diamond, column 4","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":91.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"gpqa-diamond-percent","evidenceKind":"lab_self_report","harnessId":"gpqa-diamond-percent:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEdQUUEtRGlhbW9uZC4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4gUmVhc29uaW5nOiA5ODMwNCBnZW5lcmF0aW9uIHRva2VuczsgSExFIGZ1bGwgc2V0Lg","id":"launch-kimi-k26-card-2193","ingestRunId":"launch-kimi-k26-card","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"hard_reasoning","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, GPQA-Diamond. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Reasoning: 98304 generation tokens; HLE full set.","family":"GPQA-Diamond","locator":"Evaluation Results table, GPQA-Diamond, column 5","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":94.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"mmmu-pro","evidenceKind":"lab_self_report","harnessId":"mmmu-pro:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIE1NTVUtUHJvLiBUaGlua2luZyBtb2RlOyB0ZW1wZXJhdHVyZSAxLjAsIHRvcC1wIDEuMCwgY29udGV4dCAyNjIxNDQuIENvbXBhcmF0b3Igc2V0dGluZ3M6IEdQVC01LjQgeGhpZ2gsIE9wdXM0LjYgbWF4LCBHZW1pbmkzLjFQcm8gaGlnaC4gU2VlIGJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMuIE9ubHkgb3duIEsyLjYgYW5kIHN0YXJyZWQgcmUtZXZhbHVhdGlvbnMgY2FuIGZvcm0gbmV3IGNvbXBhcmlzb25zLiBWaXNpb246IDk4MzA0IHRva2VucywgYXZlcmFnZSBvZiB0aHJlZSBydW5zOyBQeXRob24gY29uZGl0aW9uIHVzZXMgNjU1MzYgdG9rZW5zIHBlciBzdGVwLCBhdCBtb3N0NTBzdGVwcy4","id":"launch-kimi-k26-card-2194","ingestRunId":"launch-kimi-k26-card","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"multimodal","comparable":true,"configuration":"Kimi K2.6 card, MMMU-Pro. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps.","family":"MMMU-Pro","locator":"Evaluation Results table, MMMU-Pro, column 2","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":79.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"mmmu-pro","evidenceKind":"lab_self_report","harnessId":"mmmu-pro:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIE1NTVUtUHJvLiBUaGlua2luZyBtb2RlOyB0ZW1wZXJhdHVyZSAxLjAsIHRvcC1wIDEuMCwgY29udGV4dCAyNjIxNDQuIENvbXBhcmF0b3Igc2V0dGluZ3M6IEdQVC01LjQgeGhpZ2gsIE9wdXM0LjYgbWF4LCBHZW1pbmkzLjFQcm8gaGlnaC4gU2VlIGJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMuIE9ubHkgb3duIEsyLjYgYW5kIHN0YXJyZWQgcmUtZXZhbHVhdGlvbnMgY2FuIGZvcm0gbmV3IGNvbXBhcmlzb25zLiBWaXNpb246IDk4MzA0IHRva2VucywgYXZlcmFnZSBvZiB0aHJlZSBydW5zOyBQeXRob24gY29uZGl0aW9uIHVzZXMgNjU1MzYgdG9rZW5zIHBlciBzdGVwLCBhdCBtb3N0NTBzdGVwcy4","id":"launch-kimi-k26-card-2195","ingestRunId":"launch-kimi-k26-card","modelId":"gpt-5.4","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"multimodal","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, MMMU-Pro. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps.","family":"MMMU-Pro","locator":"Evaluation Results table, MMMU-Pro, column 3","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":81.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"mmmu-pro","evidenceKind":"lab_self_report","harnessId":"mmmu-pro:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIE1NTVUtUHJvLiBUaGlua2luZyBtb2RlOyB0ZW1wZXJhdHVyZSAxLjAsIHRvcC1wIDEuMCwgY29udGV4dCAyNjIxNDQuIENvbXBhcmF0b3Igc2V0dGluZ3M6IEdQVC01LjQgeGhpZ2gsIE9wdXM0LjYgbWF4LCBHZW1pbmkzLjFQcm8gaGlnaC4gU2VlIGJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMuIE9ubHkgb3duIEsyLjYgYW5kIHN0YXJyZWQgcmUtZXZhbHVhdGlvbnMgY2FuIGZvcm0gbmV3IGNvbXBhcmlzb25zLiBWaXNpb246IDk4MzA0IHRva2VucywgYXZlcmFnZSBvZiB0aHJlZSBydW5zOyBQeXRob24gY29uZGl0aW9uIHVzZXMgNjU1MzYgdG9rZW5zIHBlciBzdGVwLCBhdCBtb3N0NTBzdGVwcy4","id":"launch-kimi-k26-card-2196","ingestRunId":"launch-kimi-k26-card","modelId":"claude-opus-4-6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"multimodal","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, MMMU-Pro. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps.","family":"MMMU-Pro","locator":"Evaluation Results table, MMMU-Pro, column 4","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":73.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"mmmu-pro","evidenceKind":"lab_self_report","harnessId":"mmmu-pro:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIE1NTVUtUHJvLiBUaGlua2luZyBtb2RlOyB0ZW1wZXJhdHVyZSAxLjAsIHRvcC1wIDEuMCwgY29udGV4dCAyNjIxNDQuIENvbXBhcmF0b3Igc2V0dGluZ3M6IEdQVC01LjQgeGhpZ2gsIE9wdXM0LjYgbWF4LCBHZW1pbmkzLjFQcm8gaGlnaC4gU2VlIGJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMuIE9ubHkgb3duIEsyLjYgYW5kIHN0YXJyZWQgcmUtZXZhbHVhdGlvbnMgY2FuIGZvcm0gbmV3IGNvbXBhcmlzb25zLiBWaXNpb246IDk4MzA0IHRva2VucywgYXZlcmFnZSBvZiB0aHJlZSBydW5zOyBQeXRob24gY29uZGl0aW9uIHVzZXMgNjU1MzYgdG9rZW5zIHBlciBzdGVwLCBhdCBtb3N0NTBzdGVwcy4","id":"launch-kimi-k26-card-2197","ingestRunId":"launch-kimi-k26-card","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"multimodal","comparable":true,"configuration":"Kimi K2.6 card, MMMU-Pro. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps.","family":"MMMU-Pro","locator":"Evaluation Results table, MMMU-Pro, column 5","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":83,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"mmmu-pro-w-python","evidenceKind":"lab_self_report","harnessId":"mmmu-pro-w-python:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIE1NTVUtUHJvICh3LyBweXRob24pLiBUaGlua2luZyBtb2RlOyB0ZW1wZXJhdHVyZSAxLjAsIHRvcC1wIDEuMCwgY29udGV4dCAyNjIxNDQuIENvbXBhcmF0b3Igc2V0dGluZ3M6IEdQVC01LjQgeGhpZ2gsIE9wdXM0LjYgbWF4LCBHZW1pbmkzLjFQcm8gaGlnaC4gU2VlIGJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMuIE9ubHkgb3duIEsyLjYgYW5kIHN0YXJyZWQgcmUtZXZhbHVhdGlvbnMgY2FuIGZvcm0gbmV3IGNvbXBhcmlzb25zLiBWaXNpb246IDk4MzA0IHRva2VucywgYXZlcmFnZSBvZiB0aHJlZSBydW5zOyBQeXRob24gY29uZGl0aW9uIHVzZXMgNjU1MzYgdG9rZW5zIHBlciBzdGVwLCBhdCBtb3N0NTBzdGVwcy4","id":"launch-kimi-k26-card-2198","ingestRunId":"launch-kimi-k26-card","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"multimodal","comparable":true,"configuration":"Kimi K2.6 card, MMMU-Pro (w/ python). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps.","family":"MMMU-Pro (w/ python)","locator":"Evaluation Results table, MMMU-Pro (w/ python), column 2","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":80.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"mmmu-pro-w-python","evidenceKind":"lab_self_report","harnessId":"mmmu-pro-w-python:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIE1NTVUtUHJvICh3LyBweXRob24pLiBUaGlua2luZyBtb2RlOyB0ZW1wZXJhdHVyZSAxLjAsIHRvcC1wIDEuMCwgY29udGV4dCAyNjIxNDQuIENvbXBhcmF0b3Igc2V0dGluZ3M6IEdQVC01LjQgeGhpZ2gsIE9wdXM0LjYgbWF4LCBHZW1pbmkzLjFQcm8gaGlnaC4gU2VlIGJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMuIE9ubHkgb3duIEsyLjYgYW5kIHN0YXJyZWQgcmUtZXZhbHVhdGlvbnMgY2FuIGZvcm0gbmV3IGNvbXBhcmlzb25zLiBWaXNpb246IDk4MzA0IHRva2VucywgYXZlcmFnZSBvZiB0aHJlZSBydW5zOyBQeXRob24gY29uZGl0aW9uIHVzZXMgNjU1MzYgdG9rZW5zIHBlciBzdGVwLCBhdCBtb3N0NTBzdGVwcy4","id":"launch-kimi-k26-card-2199","ingestRunId":"launch-kimi-k26-card","modelId":"gpt-5.4","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"multimodal","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, MMMU-Pro (w/ python). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps.","family":"MMMU-Pro (w/ python)","locator":"Evaluation Results table, MMMU-Pro (w/ python), column 3","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":82.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"mmmu-pro-w-python","evidenceKind":"lab_self_report","harnessId":"mmmu-pro-w-python:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIE1NTVUtUHJvICh3LyBweXRob24pLiBUaGlua2luZyBtb2RlOyB0ZW1wZXJhdHVyZSAxLjAsIHRvcC1wIDEuMCwgY29udGV4dCAyNjIxNDQuIENvbXBhcmF0b3Igc2V0dGluZ3M6IEdQVC01LjQgeGhpZ2gsIE9wdXM0LjYgbWF4LCBHZW1pbmkzLjFQcm8gaGlnaC4gU2VlIGJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMuIE9ubHkgb3duIEsyLjYgYW5kIHN0YXJyZWQgcmUtZXZhbHVhdGlvbnMgY2FuIGZvcm0gbmV3IGNvbXBhcmlzb25zLiBWaXNpb246IDk4MzA0IHRva2VucywgYXZlcmFnZSBvZiB0aHJlZSBydW5zOyBQeXRob24gY29uZGl0aW9uIHVzZXMgNjU1MzYgdG9rZW5zIHBlciBzdGVwLCBhdCBtb3N0NTBzdGVwcy4","id":"launch-kimi-k26-card-2200","ingestRunId":"launch-kimi-k26-card","modelId":"claude-opus-4-6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"multimodal","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, MMMU-Pro (w/ python). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps.","family":"MMMU-Pro (w/ python)","locator":"Evaluation Results table, MMMU-Pro (w/ python), column 4","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":77.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"mmmu-pro-w-python","evidenceKind":"lab_self_report","harnessId":"mmmu-pro-w-python:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIE1NTVUtUHJvICh3LyBweXRob24pLiBUaGlua2luZyBtb2RlOyB0ZW1wZXJhdHVyZSAxLjAsIHRvcC1wIDEuMCwgY29udGV4dCAyNjIxNDQuIENvbXBhcmF0b3Igc2V0dGluZ3M6IEdQVC01LjQgeGhpZ2gsIE9wdXM0LjYgbWF4LCBHZW1pbmkzLjFQcm8gaGlnaC4gU2VlIGJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMuIE9ubHkgb3duIEsyLjYgYW5kIHN0YXJyZWQgcmUtZXZhbHVhdGlvbnMgY2FuIGZvcm0gbmV3IGNvbXBhcmlzb25zLiBWaXNpb246IDk4MzA0IHRva2VucywgYXZlcmFnZSBvZiB0aHJlZSBydW5zOyBQeXRob24gY29uZGl0aW9uIHVzZXMgNjU1MzYgdG9rZW5zIHBlciBzdGVwLCBhdCBtb3N0NTBzdGVwcy4","id":"launch-kimi-k26-card-2201","ingestRunId":"launch-kimi-k26-card","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"multimodal","comparable":true,"configuration":"Kimi K2.6 card, MMMU-Pro (w/ python). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps.","family":"MMMU-Pro (w/ python)","locator":"Evaluation Results table, MMMU-Pro (w/ python), column 5","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":85.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"charxiv-rq","evidenceKind":"lab_self_report","harnessId":"charxiv-rq:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIENoYXJYaXYgKFJRKS4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4gVmlzaW9uOiA5ODMwNCB0b2tlbnMsIGF2ZXJhZ2Ugb2YgdGhyZWUgcnVuczsgUHl0aG9uIGNvbmRpdGlvbiB1c2VzIDY1NTM2IHRva2VucyBwZXIgc3RlcCwgYXQgbW9zdDUwc3RlcHMu","id":"launch-kimi-k26-card-2202","ingestRunId":"launch-kimi-k26-card","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"multimodal","comparable":true,"configuration":"Kimi K2.6 card, CharXiv (RQ). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps.","family":"CharXiv (RQ)","locator":"Evaluation Results table, CharXiv (RQ), column 2","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":80.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"charxiv-rq","evidenceKind":"lab_self_report","harnessId":"charxiv-rq:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIENoYXJYaXYgKFJRKS4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4gVmlzaW9uOiA5ODMwNCB0b2tlbnMsIGF2ZXJhZ2Ugb2YgdGhyZWUgcnVuczsgUHl0aG9uIGNvbmRpdGlvbiB1c2VzIDY1NTM2IHRva2VucyBwZXIgc3RlcCwgYXQgbW9zdDUwc3RlcHMu","id":"launch-kimi-k26-card-2203","ingestRunId":"launch-kimi-k26-card","modelId":"gpt-5.4","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"multimodal","comparable":true,"configuration":"Kimi K2.6 card, CharXiv (RQ). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps.","family":"CharXiv (RQ)","locator":"Evaluation Results table, CharXiv (RQ), column 3","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":82.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"charxiv-rq","evidenceKind":"lab_self_report","harnessId":"charxiv-rq:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIENoYXJYaXYgKFJRKS4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4gVmlzaW9uOiA5ODMwNCB0b2tlbnMsIGF2ZXJhZ2Ugb2YgdGhyZWUgcnVuczsgUHl0aG9uIGNvbmRpdGlvbiB1c2VzIDY1NTM2IHRva2VucyBwZXIgc3RlcCwgYXQgbW9zdDUwc3RlcHMu","id":"launch-kimi-k26-card-2204","ingestRunId":"launch-kimi-k26-card","modelId":"claude-opus-4-6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"multimodal","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, CharXiv (RQ). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps.","family":"CharXiv (RQ)","locator":"Evaluation Results table, CharXiv (RQ), column 4","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":69.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"charxiv-rq","evidenceKind":"lab_self_report","harnessId":"charxiv-rq:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIENoYXJYaXYgKFJRKS4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4gVmlzaW9uOiA5ODMwNCB0b2tlbnMsIGF2ZXJhZ2Ugb2YgdGhyZWUgcnVuczsgUHl0aG9uIGNvbmRpdGlvbiB1c2VzIDY1NTM2IHRva2VucyBwZXIgc3RlcCwgYXQgbW9zdDUwc3RlcHMu","id":"launch-kimi-k26-card-2205","ingestRunId":"launch-kimi-k26-card","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"multimodal","comparable":true,"configuration":"Kimi K2.6 card, CharXiv (RQ). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps.","family":"CharXiv (RQ)","locator":"Evaluation Results table, CharXiv (RQ), column 5","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":80.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"charxiv-rq-w-python","evidenceKind":"lab_self_report","harnessId":"charxiv-rq-w-python:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIENoYXJYaXYgKFJRKSAody8gcHl0aG9uKS4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4gVmlzaW9uOiA5ODMwNCB0b2tlbnMsIGF2ZXJhZ2Ugb2YgdGhyZWUgcnVuczsgUHl0aG9uIGNvbmRpdGlvbiB1c2VzIDY1NTM2IHRva2VucyBwZXIgc3RlcCwgYXQgbW9zdDUwc3RlcHMu","id":"launch-kimi-k26-card-2206","ingestRunId":"launch-kimi-k26-card","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"multimodal","comparable":true,"configuration":"Kimi K2.6 card, CharXiv (RQ) (w/ python). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps.","family":"CharXiv (RQ) (w/ python)","locator":"Evaluation Results table, CharXiv (RQ) (w/ python), column 2","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":86.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"charxiv-rq-w-python","evidenceKind":"lab_self_report","harnessId":"charxiv-rq-w-python:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIENoYXJYaXYgKFJRKSAody8gcHl0aG9uKS4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4gVmlzaW9uOiA5ODMwNCB0b2tlbnMsIGF2ZXJhZ2Ugb2YgdGhyZWUgcnVuczsgUHl0aG9uIGNvbmRpdGlvbiB1c2VzIDY1NTM2IHRva2VucyBwZXIgc3RlcCwgYXQgbW9zdDUwc3RlcHMu","id":"launch-kimi-k26-card-2207","ingestRunId":"launch-kimi-k26-card","modelId":"gpt-5.4","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"multimodal","comparable":true,"configuration":"Kimi K2.6 card, CharXiv (RQ) (w/ python). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps.","family":"CharXiv (RQ) (w/ python)","locator":"Evaluation Results table, CharXiv (RQ) (w/ python), column 3","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":90,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"charxiv-rq-w-python","evidenceKind":"lab_self_report","harnessId":"charxiv-rq-w-python:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIENoYXJYaXYgKFJRKSAody8gcHl0aG9uKS4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4gVmlzaW9uOiA5ODMwNCB0b2tlbnMsIGF2ZXJhZ2Ugb2YgdGhyZWUgcnVuczsgUHl0aG9uIGNvbmRpdGlvbiB1c2VzIDY1NTM2IHRva2VucyBwZXIgc3RlcCwgYXQgbW9zdDUwc3RlcHMu","id":"launch-kimi-k26-card-2208","ingestRunId":"launch-kimi-k26-card","modelId":"claude-opus-4-6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"multimodal","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, CharXiv (RQ) (w/ python). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps.","family":"CharXiv (RQ) (w/ python)","locator":"Evaluation Results table, CharXiv (RQ) (w/ python), column 4","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":84.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"charxiv-rq-w-python","evidenceKind":"lab_self_report","harnessId":"charxiv-rq-w-python:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIENoYXJYaXYgKFJRKSAody8gcHl0aG9uKS4gVGhpbmtpbmcgbW9kZTsgdGVtcGVyYXR1cmUgMS4wLCB0b3AtcCAxLjAsIGNvbnRleHQgMjYyMTQ0LiBDb21wYXJhdG9yIHNldHRpbmdzOiBHUFQtNS40IHhoaWdoLCBPcHVzNC42IG1heCwgR2VtaW5pMy4xUHJvIGhpZ2guIFNlZSBiZW5jaG1hcmstc3BlY2lmaWMgZm9vdG5vdGVzLiBPbmx5IG93biBLMi42IGFuZCBzdGFycmVkIHJlLWV2YWx1YXRpb25zIGNhbiBmb3JtIG5ldyBjb21wYXJpc29ucy4gVmlzaW9uOiA5ODMwNCB0b2tlbnMsIGF2ZXJhZ2Ugb2YgdGhyZWUgcnVuczsgUHl0aG9uIGNvbmRpdGlvbiB1c2VzIDY1NTM2IHRva2VucyBwZXIgc3RlcCwgYXQgbW9zdDUwc3RlcHMu","id":"launch-kimi-k26-card-2209","ingestRunId":"launch-kimi-k26-card","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"multimodal","comparable":true,"configuration":"Kimi K2.6 card, CharXiv (RQ) (w/ python). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps.","family":"CharXiv (RQ) (w/ python)","locator":"Evaluation Results table, CharXiv (RQ) (w/ python), column 5","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":89.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"mathvision","evidenceKind":"lab_self_report","harnessId":"mathvision:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIE1hdGhWaXNpb24uIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMuIFZpc2lvbjogOTgzMDQgdG9rZW5zLCBhdmVyYWdlIG9mIHRocmVlIHJ1bnM7IFB5dGhvbiBjb25kaXRpb24gdXNlcyA2NTUzNiB0b2tlbnMgcGVyIHN0ZXAsIGF0IG1vc3Q1MHN0ZXBzLg","id":"launch-kimi-k26-card-2210","ingestRunId":"launch-kimi-k26-card","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"multimodal","comparable":true,"configuration":"Kimi K2.6 card, MathVision. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps.","family":"MathVision","locator":"Evaluation Results table, MathVision, column 2","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":87.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"mathvision","evidenceKind":"lab_self_report","harnessId":"mathvision:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIE1hdGhWaXNpb24uIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMuIFZpc2lvbjogOTgzMDQgdG9rZW5zLCBhdmVyYWdlIG9mIHRocmVlIHJ1bnM7IFB5dGhvbiBjb25kaXRpb24gdXNlcyA2NTUzNiB0b2tlbnMgcGVyIHN0ZXAsIGF0IG1vc3Q1MHN0ZXBzLg","id":"launch-kimi-k26-card-2211","ingestRunId":"launch-kimi-k26-card","modelId":"gpt-5.4","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"multimodal","comparable":true,"configuration":"Kimi K2.6 card, MathVision. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps.","family":"MathVision","locator":"Evaluation Results table, MathVision, column 3","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":92,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"mathvision","evidenceKind":"lab_self_report","harnessId":"mathvision:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIE1hdGhWaXNpb24uIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMuIFZpc2lvbjogOTgzMDQgdG9rZW5zLCBhdmVyYWdlIG9mIHRocmVlIHJ1bnM7IFB5dGhvbiBjb25kaXRpb24gdXNlcyA2NTUzNiB0b2tlbnMgcGVyIHN0ZXAsIGF0IG1vc3Q1MHN0ZXBzLg","id":"launch-kimi-k26-card-2212","ingestRunId":"launch-kimi-k26-card","modelId":"claude-opus-4-6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"multimodal","comparable":true,"configuration":"Kimi K2.6 card, MathVision. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps.","family":"MathVision","locator":"Evaluation Results table, MathVision, column 4","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":71.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"mathvision","evidenceKind":"lab_self_report","harnessId":"mathvision:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIE1hdGhWaXNpb24uIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMuIFZpc2lvbjogOTgzMDQgdG9rZW5zLCBhdmVyYWdlIG9mIHRocmVlIHJ1bnM7IFB5dGhvbiBjb25kaXRpb24gdXNlcyA2NTUzNiB0b2tlbnMgcGVyIHN0ZXAsIGF0IG1vc3Q1MHN0ZXBzLg","id":"launch-kimi-k26-card-2213","ingestRunId":"launch-kimi-k26-card","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"multimodal","comparable":true,"configuration":"Kimi K2.6 card, MathVision. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps.","family":"MathVision","locator":"Evaluation Results table, MathVision, column 5","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":89.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"mathvision-w-python","evidenceKind":"lab_self_report","harnessId":"mathvision-w-python:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIE1hdGhWaXNpb24gKHcvIHB5dGhvbikuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMuIFZpc2lvbjogOTgzMDQgdG9rZW5zLCBhdmVyYWdlIG9mIHRocmVlIHJ1bnM7IFB5dGhvbiBjb25kaXRpb24gdXNlcyA2NTUzNiB0b2tlbnMgcGVyIHN0ZXAsIGF0IG1vc3Q1MHN0ZXBzLg","id":"launch-kimi-k26-card-2214","ingestRunId":"launch-kimi-k26-card","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"multimodal","comparable":true,"configuration":"Kimi K2.6 card, MathVision (w/ python). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps.","family":"MathVision (w/ python)","locator":"Evaluation Results table, MathVision (w/ python), column 2","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":93.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"mathvision-w-python","evidenceKind":"lab_self_report","harnessId":"mathvision-w-python:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIE1hdGhWaXNpb24gKHcvIHB5dGhvbikuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMuIFZpc2lvbjogOTgzMDQgdG9rZW5zLCBhdmVyYWdlIG9mIHRocmVlIHJ1bnM7IFB5dGhvbiBjb25kaXRpb24gdXNlcyA2NTUzNiB0b2tlbnMgcGVyIHN0ZXAsIGF0IG1vc3Q1MHN0ZXBzLg","id":"launch-kimi-k26-card-2215","ingestRunId":"launch-kimi-k26-card","modelId":"gpt-5.4","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"multimodal","comparable":true,"configuration":"Kimi K2.6 card, MathVision (w/ python). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps.","family":"MathVision (w/ python)","locator":"Evaluation Results table, MathVision (w/ python), column 3","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":96.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"mathvision-w-python","evidenceKind":"lab_self_report","harnessId":"mathvision-w-python:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIE1hdGhWaXNpb24gKHcvIHB5dGhvbikuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMuIFZpc2lvbjogOTgzMDQgdG9rZW5zLCBhdmVyYWdlIG9mIHRocmVlIHJ1bnM7IFB5dGhvbiBjb25kaXRpb24gdXNlcyA2NTUzNiB0b2tlbnMgcGVyIHN0ZXAsIGF0IG1vc3Q1MHN0ZXBzLg","id":"launch-kimi-k26-card-2216","ingestRunId":"launch-kimi-k26-card","modelId":"claude-opus-4-6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"multimodal","comparable":true,"configuration":"Kimi K2.6 card, MathVision (w/ python). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps.","family":"MathVision (w/ python)","locator":"Evaluation Results table, MathVision (w/ python), column 4","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":84.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"mathvision-w-python","evidenceKind":"lab_self_report","harnessId":"mathvision-w-python:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIE1hdGhWaXNpb24gKHcvIHB5dGhvbikuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMuIFZpc2lvbjogOTgzMDQgdG9rZW5zLCBhdmVyYWdlIG9mIHRocmVlIHJ1bnM7IFB5dGhvbiBjb25kaXRpb24gdXNlcyA2NTUzNiB0b2tlbnMgcGVyIHN0ZXAsIGF0IG1vc3Q1MHN0ZXBzLg","id":"launch-kimi-k26-card-2217","ingestRunId":"launch-kimi-k26-card","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"multimodal","comparable":true,"configuration":"Kimi K2.6 card, MathVision (w/ python). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps.","family":"MathVision (w/ python)","locator":"Evaluation Results table, MathVision (w/ python), column 5","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":95.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"babyvision","evidenceKind":"lab_self_report","harnessId":"babyvision:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEJhYnlWaXNpb24uIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMuIFZpc2lvbjogOTgzMDQgdG9rZW5zLCBhdmVyYWdlIG9mIHRocmVlIHJ1bnM7IFB5dGhvbiBjb25kaXRpb24gdXNlcyA2NTUzNiB0b2tlbnMgcGVyIHN0ZXAsIGF0IG1vc3Q1MHN0ZXBzLg","id":"launch-kimi-k26-card-2218","ingestRunId":"launch-kimi-k26-card","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"multimodal","comparable":true,"configuration":"Kimi K2.6 card, BabyVision. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps.","family":"BabyVision","locator":"Evaluation Results table, BabyVision, column 2","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":39.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"babyvision","evidenceKind":"lab_self_report","harnessId":"babyvision:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEJhYnlWaXNpb24uIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMuIFZpc2lvbjogOTgzMDQgdG9rZW5zLCBhdmVyYWdlIG9mIHRocmVlIHJ1bnM7IFB5dGhvbiBjb25kaXRpb24gdXNlcyA2NTUzNiB0b2tlbnMgcGVyIHN0ZXAsIGF0IG1vc3Q1MHN0ZXBzLg","id":"launch-kimi-k26-card-2219","ingestRunId":"launch-kimi-k26-card","modelId":"gpt-5.4","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"multimodal","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, BabyVision. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps.","family":"BabyVision","locator":"Evaluation Results table, BabyVision, column 3","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":49.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"babyvision","evidenceKind":"lab_self_report","harnessId":"babyvision:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEJhYnlWaXNpb24uIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMuIFZpc2lvbjogOTgzMDQgdG9rZW5zLCBhdmVyYWdlIG9mIHRocmVlIHJ1bnM7IFB5dGhvbiBjb25kaXRpb24gdXNlcyA2NTUzNiB0b2tlbnMgcGVyIHN0ZXAsIGF0IG1vc3Q1MHN0ZXBzLg","id":"launch-kimi-k26-card-2220","ingestRunId":"launch-kimi-k26-card","modelId":"claude-opus-4-6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"multimodal","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, BabyVision. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps.","family":"BabyVision","locator":"Evaluation Results table, BabyVision, column 4","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":14.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"babyvision","evidenceKind":"lab_self_report","harnessId":"babyvision:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEJhYnlWaXNpb24uIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMuIFZpc2lvbjogOTgzMDQgdG9rZW5zLCBhdmVyYWdlIG9mIHRocmVlIHJ1bnM7IFB5dGhvbiBjb25kaXRpb24gdXNlcyA2NTUzNiB0b2tlbnMgcGVyIHN0ZXAsIGF0IG1vc3Q1MHN0ZXBzLg","id":"launch-kimi-k26-card-2221","ingestRunId":"launch-kimi-k26-card","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"multimodal","comparabilityReason":"Cited comparator requires original evaluation provenance; not a new matched run.","comparable":false,"configuration":"Kimi K2.6 card, BabyVision. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps.","family":"BabyVision","locator":"Evaluation Results table, BabyVision, column 5","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Cited from another official report; does not establish a new same-protocol evaluation.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":51.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"babyvision-w-python","evidenceKind":"lab_self_report","harnessId":"babyvision-w-python:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEJhYnlWaXNpb24gKHcvIHB5dGhvbikuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMuIFZpc2lvbjogOTgzMDQgdG9rZW5zLCBhdmVyYWdlIG9mIHRocmVlIHJ1bnM7IFB5dGhvbiBjb25kaXRpb24gdXNlcyA2NTUzNiB0b2tlbnMgcGVyIHN0ZXAsIGF0IG1vc3Q1MHN0ZXBzLg","id":"launch-kimi-k26-card-2222","ingestRunId":"launch-kimi-k26-card","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"multimodal","comparable":true,"configuration":"Kimi K2.6 card, BabyVision (w/ python). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps.","family":"BabyVision (w/ python)","locator":"Evaluation Results table, BabyVision (w/ python), column 2","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":68.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"babyvision-w-python","evidenceKind":"lab_self_report","harnessId":"babyvision-w-python:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEJhYnlWaXNpb24gKHcvIHB5dGhvbikuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMuIFZpc2lvbjogOTgzMDQgdG9rZW5zLCBhdmVyYWdlIG9mIHRocmVlIHJ1bnM7IFB5dGhvbiBjb25kaXRpb24gdXNlcyA2NTUzNiB0b2tlbnMgcGVyIHN0ZXAsIGF0IG1vc3Q1MHN0ZXBzLg","id":"launch-kimi-k26-card-2223","ingestRunId":"launch-kimi-k26-card","modelId":"gpt-5.4","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"multimodal","comparable":true,"configuration":"Kimi K2.6 card, BabyVision (w/ python). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps.","family":"BabyVision (w/ python)","locator":"Evaluation Results table, BabyVision (w/ python), column 3","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":80.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"babyvision-w-python","evidenceKind":"lab_self_report","harnessId":"babyvision-w-python:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEJhYnlWaXNpb24gKHcvIHB5dGhvbikuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMuIFZpc2lvbjogOTgzMDQgdG9rZW5zLCBhdmVyYWdlIG9mIHRocmVlIHJ1bnM7IFB5dGhvbiBjb25kaXRpb24gdXNlcyA2NTUzNiB0b2tlbnMgcGVyIHN0ZXAsIGF0IG1vc3Q1MHN0ZXBzLg","id":"launch-kimi-k26-card-2224","ingestRunId":"launch-kimi-k26-card","modelId":"claude-opus-4-6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"multimodal","comparable":true,"configuration":"Kimi K2.6 card, BabyVision (w/ python). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps.","family":"BabyVision (w/ python)","locator":"Evaluation Results table, BabyVision (w/ python), column 4","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":38.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"babyvision-w-python","evidenceKind":"lab_self_report","harnessId":"babyvision-w-python:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIEJhYnlWaXNpb24gKHcvIHB5dGhvbikuIFRoaW5raW5nIG1vZGU7IHRlbXBlcmF0dXJlIDEuMCwgdG9wLXAgMS4wLCBjb250ZXh0IDI2MjE0NC4gQ29tcGFyYXRvciBzZXR0aW5nczogR1BULTUuNCB4aGlnaCwgT3B1czQuNiBtYXgsIEdlbWluaTMuMVBybyBoaWdoLiBTZWUgYmVuY2htYXJrLXNwZWNpZmljIGZvb3Rub3Rlcy4gT25seSBvd24gSzIuNiBhbmQgc3RhcnJlZCByZS1ldmFsdWF0aW9ucyBjYW4gZm9ybSBuZXcgY29tcGFyaXNvbnMuIFZpc2lvbjogOTgzMDQgdG9rZW5zLCBhdmVyYWdlIG9mIHRocmVlIHJ1bnM7IFB5dGhvbiBjb25kaXRpb24gdXNlcyA2NTUzNiB0b2tlbnMgcGVyIHN0ZXAsIGF0IG1vc3Q1MHN0ZXBzLg","id":"launch-kimi-k26-card-2225","ingestRunId":"launch-kimi-k26-card","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"multimodal","comparable":true,"configuration":"Kimi K2.6 card, BabyVision (w/ python). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps.","family":"BabyVision (w/ python)","locator":"Evaluation Results table, BabyVision (w/ python), column 5","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":68.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"v-w-python","evidenceKind":"lab_self_report","harnessId":"v-w-python:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIFYqICh3LyBweXRob24pLiBUaGlua2luZyBtb2RlOyB0ZW1wZXJhdHVyZSAxLjAsIHRvcC1wIDEuMCwgY29udGV4dCAyNjIxNDQuIENvbXBhcmF0b3Igc2V0dGluZ3M6IEdQVC01LjQgeGhpZ2gsIE9wdXM0LjYgbWF4LCBHZW1pbmkzLjFQcm8gaGlnaC4gU2VlIGJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMuIE9ubHkgb3duIEsyLjYgYW5kIHN0YXJyZWQgcmUtZXZhbHVhdGlvbnMgY2FuIGZvcm0gbmV3IGNvbXBhcmlzb25zLiBWaXNpb246IDk4MzA0IHRva2VucywgYXZlcmFnZSBvZiB0aHJlZSBydW5zOyBQeXRob24gY29uZGl0aW9uIHVzZXMgNjU1MzYgdG9rZW5zIHBlciBzdGVwLCBhdCBtb3N0NTBzdGVwcy4","id":"launch-kimi-k26-card-2226","ingestRunId":"launch-kimi-k26-card","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"multimodal","comparable":true,"configuration":"Kimi K2.6 card, V* (w/ python). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps.","family":"V* (w/ python)","locator":"Evaluation Results table, V* (w/ python), column 2","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":96.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"v-w-python","evidenceKind":"lab_self_report","harnessId":"v-w-python:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIFYqICh3LyBweXRob24pLiBUaGlua2luZyBtb2RlOyB0ZW1wZXJhdHVyZSAxLjAsIHRvcC1wIDEuMCwgY29udGV4dCAyNjIxNDQuIENvbXBhcmF0b3Igc2V0dGluZ3M6IEdQVC01LjQgeGhpZ2gsIE9wdXM0LjYgbWF4LCBHZW1pbmkzLjFQcm8gaGlnaC4gU2VlIGJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMuIE9ubHkgb3duIEsyLjYgYW5kIHN0YXJyZWQgcmUtZXZhbHVhdGlvbnMgY2FuIGZvcm0gbmV3IGNvbXBhcmlzb25zLiBWaXNpb246IDk4MzA0IHRva2VucywgYXZlcmFnZSBvZiB0aHJlZSBydW5zOyBQeXRob24gY29uZGl0aW9uIHVzZXMgNjU1MzYgdG9rZW5zIHBlciBzdGVwLCBhdCBtb3N0NTBzdGVwcy4","id":"launch-kimi-k26-card-2227","ingestRunId":"launch-kimi-k26-card","modelId":"gpt-5.4","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"multimodal","comparable":true,"configuration":"Kimi K2.6 card, V* (w/ python). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps.","family":"V* (w/ python)","locator":"Evaluation Results table, V* (w/ python), column 3","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":98.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"v-w-python","evidenceKind":"lab_self_report","harnessId":"v-w-python:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIFYqICh3LyBweXRob24pLiBUaGlua2luZyBtb2RlOyB0ZW1wZXJhdHVyZSAxLjAsIHRvcC1wIDEuMCwgY29udGV4dCAyNjIxNDQuIENvbXBhcmF0b3Igc2V0dGluZ3M6IEdQVC01LjQgeGhpZ2gsIE9wdXM0LjYgbWF4LCBHZW1pbmkzLjFQcm8gaGlnaC4gU2VlIGJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMuIE9ubHkgb3duIEsyLjYgYW5kIHN0YXJyZWQgcmUtZXZhbHVhdGlvbnMgY2FuIGZvcm0gbmV3IGNvbXBhcmlzb25zLiBWaXNpb246IDk4MzA0IHRva2VucywgYXZlcmFnZSBvZiB0aHJlZSBydW5zOyBQeXRob24gY29uZGl0aW9uIHVzZXMgNjU1MzYgdG9rZW5zIHBlciBzdGVwLCBhdCBtb3N0NTBzdGVwcy4","id":"launch-kimi-k26-card-2228","ingestRunId":"launch-kimi-k26-card","modelId":"claude-opus-4-6","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"multimodal","comparable":true,"configuration":"Kimi K2.6 card, V* (w/ python). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps.","family":"V* (w/ python)","locator":"Evaluation Results table, V* (w/ python), column 4","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":86.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"v-w-python","evidenceKind":"lab_self_report","harnessId":"v-w-python:kimi-k26-card:S2ltaSBLMi42IGNhcmQsIFYqICh3LyBweXRob24pLiBUaGlua2luZyBtb2RlOyB0ZW1wZXJhdHVyZSAxLjAsIHRvcC1wIDEuMCwgY29udGV4dCAyNjIxNDQuIENvbXBhcmF0b3Igc2V0dGluZ3M6IEdQVC01LjQgeGhpZ2gsIE9wdXM0LjYgbWF4LCBHZW1pbmkzLjFQcm8gaGlnaC4gU2VlIGJlbmNobWFyay1zcGVjaWZpYyBmb290bm90ZXMuIE9ubHkgb3duIEsyLjYgYW5kIHN0YXJyZWQgcmUtZXZhbHVhdGlvbnMgY2FuIGZvcm0gbmV3IGNvbXBhcmlzb25zLiBWaXNpb246IDk4MzA0IHRva2VucywgYXZlcmFnZSBvZiB0aHJlZSBydW5zOyBQeXRob24gY29uZGl0aW9uIHVzZXMgNjU1MzYgdG9rZW5zIHBlciBzdGVwLCBhdCBtb3N0NTBzdGVwcy4","id":"launch-kimi-k26-card-2229","ingestRunId":"launch-kimi-k26-card","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-30","rawPayloadHash":"95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7","report":{"category":"multimodal","comparable":true,"configuration":"Kimi K2.6 card, V* (w/ python). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps.","family":"V* (w/ python)","locator":"Evaluation Results table, V* (w/ python), column 5","metric":"percent","notes":"Source SHA256 95db3be1d0473e482c7aa901f237ad7af8bb7929625a777b27428081bb8ea9f7. Own measurement or explicitly starred same-condition re-evaluation. Assess benchmark-specific protocol before admission.","sourceId":"kimi-k26-card","sourceTitle":"Kimi K2.6 official model card"},"score":96.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md"},{"benchmarkId":"hle-with-tools-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"hle-with-tools-source-release-snapshot-version-not-specified-pct:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1875","ingestRunId":"launch-meta","modelId":"muse-spark-1.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"Humanity's Last Exam","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":62.1,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"hle-with-tools-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"hle-with-tools-source-release-snapshot-version-not-specified-pct:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1876","ingestRunId":"launch-meta","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"Humanity's Last Exam","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":51.4,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"hle-with-tools-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"hle-with-tools-source-release-snapshot-version-not-specified-pct:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1877","ingestRunId":"launch-meta","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"Humanity's Last Exam","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":57.9,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"hle-with-tools-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"hle-with-tools-source-release-snapshot-version-not-specified-pct:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1878","ingestRunId":"launch-meta","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"Humanity's Last Exam","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":52.2,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"hle-without-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hle-without-tools-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1879","ingestRunId":"launch-meta","modelId":"muse-spark-1.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"Humanity's Last Exam","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":52.2,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"hle-without-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hle-without-tools-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1880","ingestRunId":"launch-meta","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"Humanity's Last Exam","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":45.4,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"hle-without-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hle-without-tools-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1881","ingestRunId":"launch-meta","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"Humanity's Last Exam","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":49.8,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"hle-without-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hle-without-tools-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1882","ingestRunId":"launch-meta","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"Humanity's Last Exam","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":44.8,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"mrcr-v2-1m-8-needle-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mrcr-v2-1m-8-needle-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1883","ingestRunId":"launch-meta","modelId":"muse-spark-1.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"MRCR v2 1M 8-needle","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":54.1,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"mrcr-v2-1m-8-needle-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mrcr-v2-1m-8-needle-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1884","ingestRunId":"launch-meta","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"MRCR v2 1M 8-needle","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":26.3,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"mrcr-v2-1m-8-needle-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mrcr-v2-1m-8-needle-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1885","ingestRunId":"launch-meta","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"MRCR v2 1M 8-needle","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":74,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"mcpatlas-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mcpatlas-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1886","ingestRunId":"launch-meta","modelId":"muse-spark-1.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"MCPAtlas","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":88.1,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"mcpatlas-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mcpatlas-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1887","ingestRunId":"launch-meta","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"MCPAtlas","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":78.2,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"mcpatlas-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mcpatlas-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1888","ingestRunId":"launch-meta","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"MCPAtlas","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":82.2,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"mcpatlas-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mcpatlas-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1889","ingestRunId":"launch-meta","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"MCPAtlas","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":75.3,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"toolathlon-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"toolathlon-verified-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1890","ingestRunId":"launch-meta","modelId":"muse-spark-1.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"Toolathlon Verified","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":75.6,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"toolathlon-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"toolathlon-verified-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1891","ingestRunId":"launch-meta","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"Toolathlon Verified","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":61.1,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"toolathlon-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"toolathlon-verified-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1892","ingestRunId":"launch-meta","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"Toolathlon Verified","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":76.2,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"toolathlon-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"toolathlon-verified-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1893","ingestRunId":"launch-meta","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"Toolathlon Verified","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":73.5,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"osworld-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"osworld-verified-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1894","ingestRunId":"launch-meta","modelId":"muse-spark-1.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"OSWorld","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":80.8,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"osworld-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"osworld-verified-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1895","ingestRunId":"launch-meta","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"OSWorld","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":76.2,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"osworld-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"osworld-verified-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1896","ingestRunId":"launch-meta","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"OSWorld","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":83.4,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"osworld-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"osworld-verified-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1897","ingestRunId":"launch-meta","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"OSWorld","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":78.7,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"osworld-2-0-binary-without-exec-2-0","evidenceKind":"lab_self_report","harnessId":"osworld-2-0-binary-without-exec-2-0:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1898","ingestRunId":"launch-meta","modelId":"muse-spark-1.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"OSWorld","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":14.2,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"osworld-2-0-binary-without-exec-2-0","evidenceKind":"lab_self_report","harnessId":"osworld-2-0-binary-without-exec-2-0:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1899","ingestRunId":"launch-meta","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"OSWorld","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":7.8,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"osworld-2-0-binary-without-exec-2-0","evidenceKind":"lab_self_report","harnessId":"osworld-2-0-binary-without-exec-2-0:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1900","ingestRunId":"launch-meta","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"OSWorld","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":20.6,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"osworld-2-0-binary-without-exec-2-0","evidenceKind":"lab_self_report","harnessId":"osworld-2-0-binary-without-exec-2-0:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1901","ingestRunId":"launch-meta","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"OSWorld","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":13.9,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"osworld-2-0-partial-without-exec-2-0","evidenceKind":"lab_self_report","harnessId":"osworld-2-0-partial-without-exec-2-0:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1902","ingestRunId":"launch-meta","modelId":"muse-spark-1.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"OSWorld","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":47.3,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"osworld-2-0-partial-without-exec-2-0","evidenceKind":"lab_self_report","harnessId":"osworld-2-0-partial-without-exec-2-0:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1903","ingestRunId":"launch-meta","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"OSWorld","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":30.6,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"osworld-2-0-partial-without-exec-2-0","evidenceKind":"lab_self_report","harnessId":"osworld-2-0-partial-without-exec-2-0:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1904","ingestRunId":"launch-meta","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"OSWorld","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":54.8,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"osworld-2-0-partial-without-exec-2-0","evidenceKind":"lab_self_report","harnessId":"osworld-2-0-partial-without-exec-2-0:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1905","ingestRunId":"launch-meta","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"OSWorld","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":47.5,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"webarena-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"webarena-verified-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1906","ingestRunId":"launch-meta","modelId":"muse-spark-1.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"WebArena Verified","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":69,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"webarena-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"webarena-verified-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1907","ingestRunId":"launch-meta","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"WebArena Verified","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":69,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"webarena-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"webarena-verified-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1908","ingestRunId":"launch-meta","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"WebArena Verified","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":71.2,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"webarena-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"webarena-verified-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1909","ingestRunId":"launch-meta","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"WebArena Verified","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":67,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"deepsearchqa-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"deepsearchqa-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1910","ingestRunId":"launch-meta","modelId":"muse-spark-1.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"DeepSearchQA","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":84.9,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"deepsearchqa-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"deepsearchqa-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1911","ingestRunId":"launch-meta","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"DeepSearchQA","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":71.3,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"deepsearchqa-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"deepsearchqa-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1912","ingestRunId":"launch-meta","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"DeepSearchQA","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":84.3,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"deepsearchqa-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"deepsearchqa-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1913","ingestRunId":"launch-meta","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"DeepSearchQA","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":87.8,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"gdpval-aa-v2-elo-source-release-snapshot-version-not-specified-elo","evidenceKind":"lab_self_report","harnessId":"gdpval-aa-v2-elo-source-release-snapshot-version-not-specified-elo:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1914","ingestRunId":"launch-meta","modelId":"muse-spark-1.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"GDPval-AA v2 Elo","locator":"Figure44 physical page101; methodology pages101–105","metric":"Elo","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":1381,"scoreUnit":"elo","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"gdpval-aa-v2-elo-source-release-snapshot-version-not-specified-elo","evidenceKind":"lab_self_report","harnessId":"gdpval-aa-v2-elo-source-release-snapshot-version-not-specified-elo:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1915","ingestRunId":"launch-meta","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"GDPval-AA v2 Elo","locator":"Figure44 physical page101; methodology pages101–105","metric":"Elo","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":963,"scoreUnit":"elo","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"gdpval-aa-v2-elo-source-release-snapshot-version-not-specified-elo","evidenceKind":"lab_self_report","harnessId":"gdpval-aa-v2-elo-source-release-snapshot-version-not-specified-elo:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1916","ingestRunId":"launch-meta","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"GDPval-AA v2 Elo","locator":"Figure44 physical page101; methodology pages101–105","metric":"Elo","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":1600,"scoreUnit":"elo","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"gdpval-aa-v2-elo-source-release-snapshot-version-not-specified-elo","evidenceKind":"lab_self_report","harnessId":"gdpval-aa-v2-elo-source-release-snapshot-version-not-specified-elo:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1917","ingestRunId":"launch-meta","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"GDPval-AA v2 Elo","locator":"Figure44 physical page101; methodology pages101–105","metric":"Elo","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":1494,"scoreUnit":"elo","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"jobbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"jobbench-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1918","ingestRunId":"launch-meta","modelId":"muse-spark-1.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"JobBench","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":54.7,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"jobbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"jobbench-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1919","ingestRunId":"launch-meta","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"JobBench","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":15.9,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"jobbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"jobbench-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1920","ingestRunId":"launch-meta","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"JobBench","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":48.4,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"jobbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"jobbench-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1921","ingestRunId":"launch-meta","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"JobBench","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":38.3,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"finance-agent-v2-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"finance-agent-v2-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1922","ingestRunId":"launch-meta","modelId":"muse-spark-1.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"Finance Agent v2","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":57.2,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"finance-agent-v2-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"finance-agent-v2-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1923","ingestRunId":"launch-meta","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"Finance Agent v2","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":43,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"finance-agent-v2-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"finance-agent-v2-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1924","ingestRunId":"launch-meta","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"Finance Agent v2","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":53.9,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"finance-agent-v2-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"finance-agent-v2-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1925","ingestRunId":"launch-meta","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"Finance Agent v2","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":51.8,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"terminal-bench-2-1-pct","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1-pct:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1926","ingestRunId":"launch-meta","modelId":"muse-spark-1.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"Terminal-Bench","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":80,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"terminal-bench-2-1-pct","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1-pct:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1927","ingestRunId":"launch-meta","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"Terminal-Bench","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":70.3,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"terminal-bench-2-1-pct","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1-pct:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1928","ingestRunId":"launch-meta","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"Terminal-Bench","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":82.7,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"terminal-bench-2-1-pct","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1-pct:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1929","ingestRunId":"launch-meta","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"Terminal-Bench","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":83.4,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"swe-bench-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"swe-bench-pro-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1930","ingestRunId":"launch-meta","modelId":"muse-spark-1.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"SWE-bench Pro","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":61.5,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"swe-bench-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"swe-bench-pro-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1931","ingestRunId":"launch-meta","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"SWE-bench Pro","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":54.2,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"swe-bench-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"swe-bench-pro-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1932","ingestRunId":"launch-meta","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"SWE-bench Pro","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":69.2,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"swe-bench-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"swe-bench-pro-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1933","ingestRunId":"launch-meta","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"SWE-bench Pro","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":58.6,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"deepswe-v1-1-1-1-pct","evidenceKind":"lab_self_report","harnessId":"deepswe-v1-1-1-1-pct:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1934","ingestRunId":"launch-meta","modelId":"muse-spark-1.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"DeepSWE v1.1","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":53.3,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"deepswe-v1-1-1-1-pct","evidenceKind":"lab_self_report","harnessId":"deepswe-v1-1-1-1-pct:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1935","ingestRunId":"launch-meta","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"DeepSWE v1.1","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":12,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"deepswe-v1-1-1-1-pct","evidenceKind":"lab_self_report","harnessId":"deepswe-v1-1-1-1-pct:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1936","ingestRunId":"launch-meta","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"DeepSWE v1.1","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":59,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"deepswe-v1-1-1-1-pct","evidenceKind":"lab_self_report","harnessId":"deepswe-v1-1-1-1-pct:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1937","ingestRunId":"launch-meta","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"DeepSWE v1.1","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":67,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"healthbench-professional-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"healthbench-professional-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1938","ingestRunId":"launch-meta","modelId":"muse-spark-1.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"HealthBench Professional","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":59.3,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"healthbench-professional-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"healthbench-professional-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1939","ingestRunId":"launch-meta","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"HealthBench Professional","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":41.6,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"healthbench-professional-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"healthbench-professional-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1940","ingestRunId":"launch-meta","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"HealthBench Professional","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":55.8,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"healthbench-professional-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"healthbench-professional-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1941","ingestRunId":"launch-meta","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"HealthBench Professional","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":51.8,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"charxiv-reasoning-with-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"charxiv-reasoning-with-tools-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1942","ingestRunId":"launch-meta","modelId":"muse-spark-1.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"CharXiv Reasoning with tools","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":88.4,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"charxiv-reasoning-with-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"charxiv-reasoning-with-tools-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1943","ingestRunId":"launch-meta","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"CharXiv Reasoning with tools","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":81.6,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"charxiv-reasoning-with-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"charxiv-reasoning-with-tools-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1944","ingestRunId":"launch-meta","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"CharXiv Reasoning with tools","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":89.9,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"charxiv-reasoning-with-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"charxiv-reasoning-with-tools-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1945","ingestRunId":"launch-meta","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"CharXiv Reasoning with tools","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":84.8,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"babyvision-with-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"babyvision-with-tools-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1946","ingestRunId":"launch-meta","modelId":"muse-spark-1.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"BabyVision with tools","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":76.3,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"babyvision-with-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"babyvision-with-tools-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1947","ingestRunId":"launch-meta","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"BabyVision with tools","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":51.5,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"babyvision-with-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"babyvision-with-tools-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1948","ingestRunId":"launch-meta","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"BabyVision with tools","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":81.2,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"babyvision-with-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"babyvision-with-tools-source-release-snapshot-version-not-specified:meta:TWV0YSBNb2RlbCBBUEkgeGhpZ2g7IEdlbWluaSBoaWdoOyBPcHVzIG1heDsgR1BUIHhoaWdoLiBCZXN0IG9mIHNlbGYtcmVwb3J0ZWQgb3IgTWV0YSByZXByb2R1Y3Rpb24gdW5sZXNzIHNwZWNpZmllZC4gT1NXb3JsZCBHVUkgb25seTsgVGVybWluYWwgYmFzaCBvbmx5NWF0dGVtcHRzODl0YXNrczZDUFU4R0I7IERlZXBTV0Ugbm8gaW50ZXJuZXQ1YXR0ZW1wdHMu","id":"launch-meta-1949","ingestRunId":"launch-meta","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts.","family":"BabyVision with tools","locator":"Figure44 physical page101; methodology pages101–105","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"meta","sourceTitle":"Muse Spark 1.1 evaluation report Figure44"},"score":83.6,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report"},{"benchmarkId":"gdpval-aa-2","evidenceKind":"lab_self_report","harnessId":"gdpval-aa-2:meta-muse-1-3-report:QXJ0aWZpY2lhbCBBbmFseXNpcyBTdGlycnVwIHNoZWxsL3dlYiBoYXJuZXNzOzIyMHRhc2tzOyBibGluZCBwYWlyd2lzZSBjb21wYXJpc29uczsgcmVhc29uaW5nIG1heA","id":"launch-meta-muse-1-3-report-1984","ingestRunId":"launch-meta-muse-1-3-report","modelId":"muse-spark-1.3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"Artificial Analysis Stirrup shell/web harness;220tasks; blind pairwise comparisons; reasoning max","family":"GDPval-AA","locator":"PDF page4 performance table","metric":"Elo","notes":"","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":1754,"scoreUnit":"elo","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"gdpval-aa-2","evidenceKind":"lab_self_report","harnessId":"gdpval-aa-2:meta-muse-1-3-report:QXJ0aWZpY2lhbCBBbmFseXNpcyBTdGlycnVwIHNoZWxsL3dlYiBoYXJuZXNzOzIyMHRhc2tzOyBibGluZCBwYWlyd2lzZSBjb21wYXJpc29uczsgcmVhc29uaW5nIHhoaWdo","id":"launch-meta-muse-1-3-report-1985","ingestRunId":"launch-meta-muse-1-3-report","modelId":"muse-spark-1.3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"Artificial Analysis Stirrup shell/web harness;220tasks; blind pairwise comparisons; reasoning xhigh","family":"GDPval-AA","locator":"PDF page4 performance table","metric":"Elo","notes":"","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":1709,"scoreUnit":"elo","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"gdpval-aa-2","evidenceKind":"lab_self_report","harnessId":"gdpval-aa-2:meta-muse-1-3-report:QXJ0aWZpY2lhbCBBbmFseXNpcyBTdGlycnVwIHNoZWxsL3dlYiBoYXJuZXNzOzIyMHRhc2tzOyBibGluZCBwYWlyd2lzZSBjb21wYXJpc29uczsgcmVhc29uaW5nIHhoaWdo","id":"launch-meta-muse-1-3-report-1986","ingestRunId":"launch-meta-muse-1-3-report","modelId":"muse-spark-1.2","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"Artificial Analysis Stirrup shell/web harness;220tasks; blind pairwise comparisons; reasoning xhigh","family":"GDPval-AA","locator":"PDF page4 performance table","metric":"Elo","notes":"","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":1615,"scoreUnit":"elo","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"gdpval-aa-2","evidenceKind":"lab_self_report","harnessId":"gdpval-aa-2:meta-muse-1-3-report:QXJ0aWZpY2lhbCBBbmFseXNpcyBTdGlycnVwIHNoZWxsL3dlYiBoYXJuZXNzOzIyMHRhc2tzOyBibGluZCBwYWlyd2lzZSBjb21wYXJpc29uczsgcmVhc29uaW5nIG1heA","id":"launch-meta-muse-1-3-report-1987","ingestRunId":"launch-meta-muse-1-3-report","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"Artificial Analysis Stirrup shell/web harness;220tasks; blind pairwise comparisons; reasoning max","family":"GDPval-AA","locator":"PDF page4 performance table","metric":"Elo","notes":"","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":1710,"scoreUnit":"elo","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"gdpval-aa-2","evidenceKind":"lab_self_report","harnessId":"gdpval-aa-2:meta-muse-1-3-report:QXJ0aWZpY2lhbCBBbmFseXNpcyBTdGlycnVwIHNoZWxsL3dlYiBoYXJuZXNzOzIyMHRhc2tzOyBibGluZCBwYWlyd2lzZSBjb21wYXJpc29uczsgcmVhc29uaW5nIG1heA","id":"launch-meta-muse-1-3-report-1988","ingestRunId":"launch-meta-muse-1-3-report","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"Artificial Analysis Stirrup shell/web harness;220tasks; blind pairwise comparisons; reasoning max","family":"GDPval-AA","locator":"PDF page4 performance table","metric":"Elo","notes":"","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":1824,"scoreUnit":"elo","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"jobbench","evidenceKind":"lab_self_report","harnessId":"jobbench:meta-muse-1-3-report:NjV0YXNrczsgbWVhbiBydWJyaWMgc2NvcmU7IG9mZmljaWFsIE9wZW5Db2RlIGhhcm5lc3MgYW5kIGZpbGUtYXdhcmUgZ3JhZGVyOyByZWFzb25pbmcgbWF4","id":"launch-meta-muse-1-3-report-1989","ingestRunId":"launch-meta-muse-1-3-report","modelId":"muse-spark-1.3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"65tasks; mean rubric score; official OpenCode harness and file-aware grader; reasoning max","family":"JobBench","locator":"PDF page4 performance table","metric":"percent","notes":"","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":64.9,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"jobbench","evidenceKind":"lab_self_report","harnessId":"jobbench:meta-muse-1-3-report:NjV0YXNrczsgbWVhbiBydWJyaWMgc2NvcmU7IG9mZmljaWFsIE9wZW5Db2RlIGhhcm5lc3MgYW5kIGZpbGUtYXdhcmUgZ3JhZGVyOyByZWFzb25pbmcgeGhpZ2g","id":"launch-meta-muse-1-3-report-1990","ingestRunId":"launch-meta-muse-1-3-report","modelId":"muse-spark-1.3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"65tasks; mean rubric score; official OpenCode harness and file-aware grader; reasoning xhigh","family":"JobBench","locator":"PDF page4 performance table","metric":"percent","notes":"","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":61.2,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"jobbench","evidenceKind":"lab_self_report","harnessId":"jobbench:meta-muse-1-3-report:NjV0YXNrczsgbWVhbiBydWJyaWMgc2NvcmU7IG9mZmljaWFsIE9wZW5Db2RlIGhhcm5lc3MgYW5kIGZpbGUtYXdhcmUgZ3JhZGVyOyByZWFzb25pbmcgeGhpZ2g","id":"launch-meta-muse-1-3-report-1991","ingestRunId":"launch-meta-muse-1-3-report","modelId":"muse-spark-1.2","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"65tasks; mean rubric score; official OpenCode harness and file-aware grader; reasoning xhigh","family":"JobBench","locator":"PDF page4 performance table","metric":"percent","notes":"","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":61.6,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"jobbench","evidenceKind":"lab_self_report","harnessId":"jobbench:meta-muse-1-3-report:NjV0YXNrczsgbWVhbiBydWJyaWMgc2NvcmU7IG9mZmljaWFsIE9wZW5Db2RlIGhhcm5lc3MgYW5kIGZpbGUtYXdhcmUgZ3JhZGVyOyByZWFzb25pbmcgbWF4","id":"launch-meta-muse-1-3-report-1992","ingestRunId":"launch-meta-muse-1-3-report","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"65tasks; mean rubric score; official OpenCode harness and file-aware grader; reasoning max","family":"JobBench","locator":"PDF page4 performance table","metric":"percent","notes":"","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":45.4,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"jobbench","evidenceKind":"lab_self_report","harnessId":"jobbench:meta-muse-1-3-report:NjV0YXNrczsgbWVhbiBydWJyaWMgc2NvcmU7IG9mZmljaWFsIE9wZW5Db2RlIGhhcm5lc3MgYW5kIGZpbGUtYXdhcmUgZ3JhZGVyOyByZWFzb25pbmcgbWF4","id":"launch-meta-muse-1-3-report-1993","ingestRunId":"launch-meta-muse-1-3-report","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"65tasks; mean rubric score; official OpenCode harness and file-aware grader; reasoning max","family":"JobBench","locator":"PDF page4 performance table","metric":"percent","notes":"","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":65.7,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"osworld-partial-2-0-08-08","evidenceKind":"lab_self_report","harnessId":"osworld-partial-2-0-08-08:meta-muse-1-3-report:MTA4dGFza3M7IGNvbW1vbiBpbnRlcm5hbCBHVUkgZnJhbWV3b3JrOyBleGVjdXRpb24tYmFzZWQgY2hlY2tlcnM7IHBhcnRpYWwgbWV0cmljOyByZWFzb25pbmcgbWF4","id":"launch-meta-muse-1-3-report-1994","ingestRunId":"launch-meta-muse-1-3-report","modelId":"muse-spark-1.3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"108tasks; common internal GUI framework; execution-based checkers; partial metric; reasoning max","family":"OSWorld","locator":"PDF page4 performance table","metric":"percent","notes":"Muse1.2 uses older06.24tasks; all others08.08. Do not compare these versions as identical.","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":66.9,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"osworld-partial-2-0-08-08","evidenceKind":"lab_self_report","harnessId":"osworld-partial-2-0-08-08:meta-muse-1-3-report:MTA4dGFza3M7IGNvbW1vbiBpbnRlcm5hbCBHVUkgZnJhbWV3b3JrOyBleGVjdXRpb24tYmFzZWQgY2hlY2tlcnM7IHBhcnRpYWwgbWV0cmljOyByZWFzb25pbmcgeGhpZ2g","id":"launch-meta-muse-1-3-report-1995","ingestRunId":"launch-meta-muse-1-3-report","modelId":"muse-spark-1.3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"108tasks; common internal GUI framework; execution-based checkers; partial metric; reasoning xhigh","family":"OSWorld","locator":"PDF page4 performance table","metric":"percent","notes":"Muse1.2 uses older06.24tasks; all others08.08. Do not compare these versions as identical.","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":59,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"osworld-partial-2-0-06-24","evidenceKind":"lab_self_report","harnessId":"osworld-partial-2-0-06-24:meta-muse-1-3-report:MTA4dGFza3M7IGNvbW1vbiBpbnRlcm5hbCBHVUkgZnJhbWV3b3JrOyBleGVjdXRpb24tYmFzZWQgY2hlY2tlcnM7IHBhcnRpYWwgbWV0cmljOyByZWFzb25pbmcgeGhpZ2g","id":"launch-meta-muse-1-3-report-1996","ingestRunId":"launch-meta-muse-1-3-report","modelId":"muse-spark-1.2","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"108tasks; common internal GUI framework; execution-based checkers; partial metric; reasoning xhigh","family":"OSWorld","locator":"PDF page4 performance table","metric":"percent","notes":"Muse1.2 uses older06.24tasks; all others08.08. Do not compare these versions as identical.","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":47.6,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"osworld-partial-2-0-08-08","evidenceKind":"lab_self_report","harnessId":"osworld-partial-2-0-08-08:meta-muse-1-3-report:MTA4dGFza3M7IGNvbW1vbiBpbnRlcm5hbCBHVUkgZnJhbWV3b3JrOyBleGVjdXRpb24tYmFzZWQgY2hlY2tlcnM7IHBhcnRpYWwgbWV0cmljOyByZWFzb25pbmcgbWF4","id":"launch-meta-muse-1-3-report-1997","ingestRunId":"launch-meta-muse-1-3-report","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"108tasks; common internal GUI framework; execution-based checkers; partial metric; reasoning max","family":"OSWorld","locator":"PDF page4 performance table","metric":"percent","notes":"Muse1.2 uses older06.24tasks; all others08.08. Do not compare these versions as identical.","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":62.7,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"osworld-partial-2-0-08-08","evidenceKind":"lab_self_report","harnessId":"osworld-partial-2-0-08-08:meta-muse-1-3-report:MTA4dGFza3M7IGNvbW1vbiBpbnRlcm5hbCBHVUkgZnJhbWV3b3JrOyBleGVjdXRpb24tYmFzZWQgY2hlY2tlcnM7IHBhcnRpYWwgbWV0cmljOyByZWFzb25pbmcgbWF4","id":"launch-meta-muse-1-3-report-1998","ingestRunId":"launch-meta-muse-1-3-report","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"108tasks; common internal GUI framework; execution-based checkers; partial metric; reasoning max","family":"OSWorld","locator":"PDF page4 performance table","metric":"percent","notes":"Muse1.2 uses older06.24tasks; all others08.08. Do not compare these versions as identical.","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":68.3,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"osworld-binary-2-0-08-08","evidenceKind":"lab_self_report","harnessId":"osworld-binary-2-0-08-08:meta-muse-1-3-report:MTA4dGFza3M7IGNvbW1vbiBpbnRlcm5hbCBHVUkgZnJhbWV3b3JrOyBleGVjdXRpb24tYmFzZWQgY2hlY2tlcnM7IGJpbmFyeSBtZXRyaWM7IHJlYXNvbmluZyBtYXg","id":"launch-meta-muse-1-3-report-1999","ingestRunId":"launch-meta-muse-1-3-report","modelId":"muse-spark-1.3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"108tasks; common internal GUI framework; execution-based checkers; binary metric; reasoning max","family":"OSWorld","locator":"PDF page4 performance table","metric":"percent","notes":"Muse1.2 uses older06.24tasks; all others08.08. Do not compare these versions as identical.","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":32,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"osworld-binary-2-0-08-08","evidenceKind":"lab_self_report","harnessId":"osworld-binary-2-0-08-08:meta-muse-1-3-report:MTA4dGFza3M7IGNvbW1vbiBpbnRlcm5hbCBHVUkgZnJhbWV3b3JrOyBleGVjdXRpb24tYmFzZWQgY2hlY2tlcnM7IGJpbmFyeSBtZXRyaWM7IHJlYXNvbmluZyB4aGlnaA","id":"launch-meta-muse-1-3-report-2000","ingestRunId":"launch-meta-muse-1-3-report","modelId":"muse-spark-1.3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"108tasks; common internal GUI framework; execution-based checkers; binary metric; reasoning xhigh","family":"OSWorld","locator":"PDF page4 performance table","metric":"percent","notes":"Muse1.2 uses older06.24tasks; all others08.08. Do not compare these versions as identical.","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":26.7,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"osworld-binary-2-0-06-24","evidenceKind":"lab_self_report","harnessId":"osworld-binary-2-0-06-24:meta-muse-1-3-report:MTA4dGFza3M7IGNvbW1vbiBpbnRlcm5hbCBHVUkgZnJhbWV3b3JrOyBleGVjdXRpb24tYmFzZWQgY2hlY2tlcnM7IGJpbmFyeSBtZXRyaWM7IHJlYXNvbmluZyB4aGlnaA","id":"launch-meta-muse-1-3-report-2001","ingestRunId":"launch-meta-muse-1-3-report","modelId":"muse-spark-1.2","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"108tasks; common internal GUI framework; execution-based checkers; binary metric; reasoning xhigh","family":"OSWorld","locator":"PDF page4 performance table","metric":"percent","notes":"Muse1.2 uses older06.24tasks; all others08.08. Do not compare these versions as identical.","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":17.9,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"osworld-binary-2-0-08-08","evidenceKind":"lab_self_report","harnessId":"osworld-binary-2-0-08-08:meta-muse-1-3-report:MTA4dGFza3M7IGNvbW1vbiBpbnRlcm5hbCBHVUkgZnJhbWV3b3JrOyBleGVjdXRpb24tYmFzZWQgY2hlY2tlcnM7IGJpbmFyeSBtZXRyaWM7IHJlYXNvbmluZyBtYXg","id":"launch-meta-muse-1-3-report-2002","ingestRunId":"launch-meta-muse-1-3-report","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"108tasks; common internal GUI framework; execution-based checkers; binary metric; reasoning max","family":"OSWorld","locator":"PDF page4 performance table","metric":"percent","notes":"Muse1.2 uses older06.24tasks; all others08.08. Do not compare these versions as identical.","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":27.3,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"osworld-binary-2-0-08-08","evidenceKind":"lab_self_report","harnessId":"osworld-binary-2-0-08-08:meta-muse-1-3-report:MTA4dGFza3M7IGNvbW1vbiBpbnRlcm5hbCBHVUkgZnJhbWV3b3JrOyBleGVjdXRpb24tYmFzZWQgY2hlY2tlcnM7IGJpbmFyeSBtZXRyaWM7IHJlYXNvbmluZyBtYXg","id":"launch-meta-muse-1-3-report-2003","ingestRunId":"launch-meta-muse-1-3-report","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"108tasks; common internal GUI framework; execution-based checkers; binary metric; reasoning max","family":"OSWorld","locator":"PDF page4 performance table","metric":"percent","notes":"Muse1.2 uses older06.24tasks; all others08.08. Do not compare these versions as identical.","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":31.4,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"deepsearchqa","evidenceKind":"lab_self_report","harnessId":"deepsearchqa:meta-muse-1-3-report:OTAwcXVlc3Rpb25zOyBjb21tb24gc2VhcmNoIGJhY2tlbmQvYnJvd3NlciBoYXJuZXNzOyBhbnN3ZXItc2V0IEYxOyByZWFzb25pbmcgbWF4","id":"launch-meta-muse-1-3-report-2004","ingestRunId":"launch-meta-muse-1-3-report","modelId":"muse-spark-1.3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"900questions; common search backend/browser harness; answer-set F1; reasoning max","family":"DeepSearchQA","locator":"PDF page4 performance table","metric":"percent F1","notes":"","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":90.3,"scoreUnit":"index","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"deepsearchqa","evidenceKind":"lab_self_report","harnessId":"deepsearchqa:meta-muse-1-3-report:OTAwcXVlc3Rpb25zOyBjb21tb24gc2VhcmNoIGJhY2tlbmQvYnJvd3NlciBoYXJuZXNzOyBhbnN3ZXItc2V0IEYxOyByZWFzb25pbmcgeGhpZ2g","id":"launch-meta-muse-1-3-report-2005","ingestRunId":"launch-meta-muse-1-3-report","modelId":"muse-spark-1.3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"900questions; common search backend/browser harness; answer-set F1; reasoning xhigh","family":"DeepSearchQA","locator":"PDF page4 performance table","metric":"percent F1","notes":"","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":89.4,"scoreUnit":"index","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"deepsearchqa","evidenceKind":"lab_self_report","harnessId":"deepsearchqa:meta-muse-1-3-report:OTAwcXVlc3Rpb25zOyBjb21tb24gc2VhcmNoIGJhY2tlbmQvYnJvd3NlciBoYXJuZXNzOyBhbnN3ZXItc2V0IEYxOyByZWFzb25pbmcgeGhpZ2g","id":"launch-meta-muse-1-3-report-2006","ingestRunId":"launch-meta-muse-1-3-report","modelId":"muse-spark-1.2","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"900questions; common search backend/browser harness; answer-set F1; reasoning xhigh","family":"DeepSearchQA","locator":"PDF page4 performance table","metric":"percent F1","notes":"","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":85.9,"scoreUnit":"index","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"deepsearchqa","evidenceKind":"lab_self_report","harnessId":"deepsearchqa:meta-muse-1-3-report:OTAwcXVlc3Rpb25zOyBjb21tb24gc2VhcmNoIGJhY2tlbmQvYnJvd3NlciBoYXJuZXNzOyBhbnN3ZXItc2V0IEYxOyByZWFzb25pbmcgbWF4","id":"launch-meta-muse-1-3-report-2007","ingestRunId":"launch-meta-muse-1-3-report","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"900questions; common search backend/browser harness; answer-set F1; reasoning max","family":"DeepSearchQA","locator":"PDF page4 performance table","metric":"percent F1","notes":"","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":93.1,"scoreUnit":"index","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"deepsearchqa","evidenceKind":"lab_self_report","harnessId":"deepsearchqa:meta-muse-1-3-report:OTAwcXVlc3Rpb25zOyBjb21tb24gc2VhcmNoIGJhY2tlbmQvYnJvd3NlciBoYXJuZXNzOyBhbnN3ZXItc2V0IEYxOyByZWFzb25pbmcgbWF4","id":"launch-meta-muse-1-3-report-2008","ingestRunId":"launch-meta-muse-1-3-report","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"900questions; common search backend/browser harness; answer-set F1; reasoning max","family":"DeepSearchQA","locator":"PDF page4 performance table","metric":"percent F1","notes":"","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":90.4,"scoreUnit":"index","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"agentic-if-index-internal","evidenceKind":"lab_self_report","harnessId":"agentic-if-index-internal:meta-muse-1-3-report:SW50ZXJuYWwgY29tcG9zaXRlIGluc3RydWN0aW9uLWZvbGxvd2luZyBldmFsdWF0aW9uczsgbm8gZml4ZWQgdGFzayBjb3VudDsgcmVhc29uaW5nIG1heA","id":"launch-meta-muse-1-3-report-2009","ingestRunId":"launch-meta-muse-1-3-report","modelId":"muse-spark-1.3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Internal composite instruction-following evaluations; no fixed task count; reasoning max","family":"Agentic IF Index","locator":"PDF page4 performance table","metric":"index","notes":"","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":57.8,"scoreUnit":"index","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"agentic-if-index-internal","evidenceKind":"lab_self_report","harnessId":"agentic-if-index-internal:meta-muse-1-3-report:SW50ZXJuYWwgY29tcG9zaXRlIGluc3RydWN0aW9uLWZvbGxvd2luZyBldmFsdWF0aW9uczsgbm8gZml4ZWQgdGFzayBjb3VudDsgcmVhc29uaW5nIHhoaWdo","id":"launch-meta-muse-1-3-report-2010","ingestRunId":"launch-meta-muse-1-3-report","modelId":"muse-spark-1.3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Internal composite instruction-following evaluations; no fixed task count; reasoning xhigh","family":"Agentic IF Index","locator":"PDF page4 performance table","metric":"index","notes":"","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":55.7,"scoreUnit":"index","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"agentic-if-index-internal","evidenceKind":"lab_self_report","harnessId":"agentic-if-index-internal:meta-muse-1-3-report:SW50ZXJuYWwgY29tcG9zaXRlIGluc3RydWN0aW9uLWZvbGxvd2luZyBldmFsdWF0aW9uczsgbm8gZml4ZWQgdGFzayBjb3VudDsgcmVhc29uaW5nIHhoaWdo","id":"launch-meta-muse-1-3-report-2011","ingestRunId":"launch-meta-muse-1-3-report","modelId":"muse-spark-1.2","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Internal composite instruction-following evaluations; no fixed task count; reasoning xhigh","family":"Agentic IF Index","locator":"PDF page4 performance table","metric":"index","notes":"","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":46.2,"scoreUnit":"index","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"agentic-if-index-internal","evidenceKind":"lab_self_report","harnessId":"agentic-if-index-internal:meta-muse-1-3-report:SW50ZXJuYWwgY29tcG9zaXRlIGluc3RydWN0aW9uLWZvbGxvd2luZyBldmFsdWF0aW9uczsgbm8gZml4ZWQgdGFzayBjb3VudDsgcmVhc29uaW5nIG1heA","id":"launch-meta-muse-1-3-report-2012","ingestRunId":"launch-meta-muse-1-3-report","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Internal composite instruction-following evaluations; no fixed task count; reasoning max","family":"Agentic IF Index","locator":"PDF page4 performance table","metric":"index","notes":"","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":60.5,"scoreUnit":"index","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"agentic-if-index-internal","evidenceKind":"lab_self_report","harnessId":"agentic-if-index-internal:meta-muse-1-3-report:SW50ZXJuYWwgY29tcG9zaXRlIGluc3RydWN0aW9uLWZvbGxvd2luZyBldmFsdWF0aW9uczsgbm8gZml4ZWQgdGFzayBjb3VudDsgcmVhc29uaW5nIG1heA","id":"launch-meta-muse-1-3-report-2013","ingestRunId":"launch-meta-muse-1-3-report","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Internal composite instruction-following evaluations; no fixed task count; reasoning max","family":"Agentic IF Index","locator":"PDF page4 performance table","metric":"index","notes":"","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":59.1,"scoreUnit":"index","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"automationbench-public-v3","evidenceKind":"lab_self_report","harnessId":"automationbench-public-v3:meta-muse-1-3-report:NjAwcHVibGljIHdvcmtmbG93IHRhc2tzOyBkZXRlcm1pbmlzdGljIGVuZC1zdGF0ZSBhc3NlcnRpb25zO3Bhc3NAMTsgcmVhc29uaW5nIG1heA","id":"launch-meta-muse-1-3-report-2014","ingestRunId":"launch-meta-muse-1-3-report","modelId":"muse-spark-1.3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"600public workflow tasks; deterministic end-state assertions;pass@1; reasoning max","family":"AutomationBench","locator":"PDF page4 performance table","metric":"percent","notes":"","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":49.6,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"automationbench-public-v3","evidenceKind":"lab_self_report","harnessId":"automationbench-public-v3:meta-muse-1-3-report:NjAwcHVibGljIHdvcmtmbG93IHRhc2tzOyBkZXRlcm1pbmlzdGljIGVuZC1zdGF0ZSBhc3NlcnRpb25zO3Bhc3NAMTsgcmVhc29uaW5nIHhoaWdo","id":"launch-meta-muse-1-3-report-2015","ingestRunId":"launch-meta-muse-1-3-report","modelId":"muse-spark-1.3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"600public workflow tasks; deterministic end-state assertions;pass@1; reasoning xhigh","family":"AutomationBench","locator":"PDF page4 performance table","metric":"percent","notes":"","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":48.3,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"automationbench-public-v3","evidenceKind":"lab_self_report","harnessId":"automationbench-public-v3:meta-muse-1-3-report:NjAwcHVibGljIHdvcmtmbG93IHRhc2tzOyBkZXRlcm1pbmlzdGljIGVuZC1zdGF0ZSBhc3NlcnRpb25zO3Bhc3NAMTsgcmVhc29uaW5nIHhoaWdo","id":"launch-meta-muse-1-3-report-2016","ingestRunId":"launch-meta-muse-1-3-report","modelId":"muse-spark-1.2","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"600public workflow tasks; deterministic end-state assertions;pass@1; reasoning xhigh","family":"AutomationBench","locator":"PDF page4 performance table","metric":"percent","notes":"","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":38.2,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"automationbench-public-v3","evidenceKind":"lab_self_report","harnessId":"automationbench-public-v3:meta-muse-1-3-report:NjAwcHVibGljIHdvcmtmbG93IHRhc2tzOyBkZXRlcm1pbmlzdGljIGVuZC1zdGF0ZSBhc3NlcnRpb25zO3Bhc3NAMTsgcmVhc29uaW5nIG1heA","id":"launch-meta-muse-1-3-report-2017","ingestRunId":"launch-meta-muse-1-3-report","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"600public workflow tasks; deterministic end-state assertions;pass@1; reasoning max","family":"AutomationBench","locator":"PDF page4 performance table","metric":"percent","notes":"","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":46.7,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"automationbench-public-v3","evidenceKind":"lab_self_report","harnessId":"automationbench-public-v3:meta-muse-1-3-report:NjAwcHVibGljIHdvcmtmbG93IHRhc2tzOyBkZXRlcm1pbmlzdGljIGVuZC1zdGF0ZSBhc3NlcnRpb25zO3Bhc3NAMTsgcmVhc29uaW5nIG1heA","id":"launch-meta-muse-1-3-report-2018","ingestRunId":"launch-meta-muse-1-3-report","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"600public workflow tasks; deterministic end-state assertions;pass@1; reasoning max","family":"AutomationBench","locator":"PDF page4 performance table","metric":"percent","notes":"","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":50.3,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"mrcr-2-256k-512k","evidenceKind":"lab_self_report","harnessId":"mrcr-2-256k-512k:meta-muse-1-3-report:OG5lZWRsZTsxMDBleGFtcGxlcy9iYW5kIHJlYmlubmVkIGJ5bzIwMGtfYmFzZTsgbm8gdG9vbHM7IHNlcXVlbmNlLW1hdGNoZXIgcmF0aW87IEdQVCBmcm9tT3BlbkFJY2FyZDsgcmVhc29uaW5nIG1heA","id":"launch-meta-muse-1-3-report-2019","ingestRunId":"launch-meta-muse-1-3-report","modelId":"muse-spark-1.3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"8needle;100examples/band rebinned byo200k_base; no tools; sequence-matcher ratio; GPT fromOpenAIcard; reasoning max","family":"MRCR","locator":"PDF page4 performance table","metric":"percent sequence match","notes":"","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":98.5,"scoreUnit":"index","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"mrcr-2-256k-512k","evidenceKind":"lab_self_report","harnessId":"mrcr-2-256k-512k:meta-muse-1-3-report:OG5lZWRsZTsxMDBleGFtcGxlcy9iYW5kIHJlYmlubmVkIGJ5bzIwMGtfYmFzZTsgbm8gdG9vbHM7IHNlcXVlbmNlLW1hdGNoZXIgcmF0aW87IEdQVCBmcm9tT3BlbkFJY2FyZDsgcmVhc29uaW5nIHhoaWdo","id":"launch-meta-muse-1-3-report-2020","ingestRunId":"launch-meta-muse-1-3-report","modelId":"muse-spark-1.3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"8needle;100examples/band rebinned byo200k_base; no tools; sequence-matcher ratio; GPT fromOpenAIcard; reasoning xhigh","family":"MRCR","locator":"PDF page4 performance table","metric":"percent sequence match","notes":"","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":97.6,"scoreUnit":"index","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"mrcr-2-256k-512k","evidenceKind":"lab_self_report","harnessId":"mrcr-2-256k-512k:meta-muse-1-3-report:OG5lZWRsZTsxMDBleGFtcGxlcy9iYW5kIHJlYmlubmVkIGJ5bzIwMGtfYmFzZTsgbm8gdG9vbHM7IHNlcXVlbmNlLW1hdGNoZXIgcmF0aW87IEdQVCBmcm9tT3BlbkFJY2FyZDsgcmVhc29uaW5nIHhoaWdo","id":"launch-meta-muse-1-3-report-2021","ingestRunId":"launch-meta-muse-1-3-report","modelId":"muse-spark-1.2","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"8needle;100examples/band rebinned byo200k_base; no tools; sequence-matcher ratio; GPT fromOpenAIcard; reasoning xhigh","family":"MRCR","locator":"PDF page4 performance table","metric":"percent sequence match","notes":"","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":66.3,"scoreUnit":"index","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"mrcr-2-256k-512k","evidenceKind":"lab_self_report","harnessId":"mrcr-2-256k-512k:meta-muse-1-3-report:OG5lZWRsZTsxMDBleGFtcGxlcy9iYW5kIHJlYmlubmVkIGJ5bzIwMGtfYmFzZTsgbm8gdG9vbHM7IHNlcXVlbmNlLW1hdGNoZXIgcmF0aW87IEdQVCBmcm9tT3BlbkFJY2FyZDsgcmVhc29uaW5nIG1heA","id":"launch-meta-muse-1-3-report-2022","ingestRunId":"launch-meta-muse-1-3-report","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"8needle;100examples/band rebinned byo200k_base; no tools; sequence-matcher ratio; GPT fromOpenAIcard; reasoning max","family":"MRCR","locator":"PDF page4 performance table","metric":"percent sequence match","notes":"","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":91.5,"scoreUnit":"index","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"mrcr-2-512k-1m","evidenceKind":"lab_self_report","harnessId":"mrcr-2-512k-1m:meta-muse-1-3-report:OG5lZWRsZTsxMDBleGFtcGxlcy9iYW5kIHJlYmlubmVkIGJ5bzIwMGtfYmFzZTsgbm8gdG9vbHM7IHNlcXVlbmNlLW1hdGNoZXIgcmF0aW87IEdQVCBmcm9tT3BlbkFJY2FyZDsgcmVhc29uaW5nIG1heA","id":"launch-meta-muse-1-3-report-2023","ingestRunId":"launch-meta-muse-1-3-report","modelId":"muse-spark-1.3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"8needle;100examples/band rebinned byo200k_base; no tools; sequence-matcher ratio; GPT fromOpenAIcard; reasoning max","family":"MRCR","locator":"PDF page4 performance table","metric":"percent sequence match","notes":"","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":98.1,"scoreUnit":"index","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"mrcr-2-512k-1m","evidenceKind":"lab_self_report","harnessId":"mrcr-2-512k-1m:meta-muse-1-3-report:OG5lZWRsZTsxMDBleGFtcGxlcy9iYW5kIHJlYmlubmVkIGJ5bzIwMGtfYmFzZTsgbm8gdG9vbHM7IHNlcXVlbmNlLW1hdGNoZXIgcmF0aW87IEdQVCBmcm9tT3BlbkFJY2FyZDsgcmVhc29uaW5nIHhoaWdo","id":"launch-meta-muse-1-3-report-2024","ingestRunId":"launch-meta-muse-1-3-report","modelId":"muse-spark-1.3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"8needle;100examples/band rebinned byo200k_base; no tools; sequence-matcher ratio; GPT fromOpenAIcard; reasoning xhigh","family":"MRCR","locator":"PDF page4 performance table","metric":"percent sequence match","notes":"","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":93.1,"scoreUnit":"index","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"mrcr-2-512k-1m","evidenceKind":"lab_self_report","harnessId":"mrcr-2-512k-1m:meta-muse-1-3-report:OG5lZWRsZTsxMDBleGFtcGxlcy9iYW5kIHJlYmlubmVkIGJ5bzIwMGtfYmFzZTsgbm8gdG9vbHM7IHNlcXVlbmNlLW1hdGNoZXIgcmF0aW87IEdQVCBmcm9tT3BlbkFJY2FyZDsgcmVhc29uaW5nIHhoaWdo","id":"launch-meta-muse-1-3-report-2025","ingestRunId":"launch-meta-muse-1-3-report","modelId":"muse-spark-1.2","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"8needle;100examples/band rebinned byo200k_base; no tools; sequence-matcher ratio; GPT fromOpenAIcard; reasoning xhigh","family":"MRCR","locator":"PDF page4 performance table","metric":"percent sequence match","notes":"","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":55.5,"scoreUnit":"index","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"mrcr-2-512k-1m","evidenceKind":"lab_self_report","harnessId":"mrcr-2-512k-1m:meta-muse-1-3-report:OG5lZWRsZTsxMDBleGFtcGxlcy9iYW5kIHJlYmlubmVkIGJ5bzIwMGtfYmFzZTsgbm8gdG9vbHM7IHNlcXVlbmNlLW1hdGNoZXIgcmF0aW87IEdQVCBmcm9tT3BlbkFJY2FyZDsgcmVhc29uaW5nIG1heA","id":"launch-meta-muse-1-3-report-2026","ingestRunId":"launch-meta-muse-1-3-report","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"8needle;100examples/band rebinned byo200k_base; no tools; sequence-matcher ratio; GPT fromOpenAIcard; reasoning max","family":"MRCR","locator":"PDF page4 performance table","metric":"percent sequence match","notes":"","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":73.8,"scoreUnit":"index","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"deepswe-1-1-percent","evidenceKind":"lab_self_report","harnessId":"deepswe-1-1-percent:meta-muse-1-3-report:MTEzdGFza3M7IE11c2UxLjNtaW5pLXN3ZS1hZ2VudDsgY29tcGFyYXRvcnMgb2ZmaWNpYWxEYXRhY3VydmUgYm9hcmQ7IHJlYXNvbmluZyBtYXg","id":"launch-meta-muse-1-3-report-2027","ingestRunId":"launch-meta-muse-1-3-report","modelId":"muse-spark-1.3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"113tasks; Muse1.3mini-swe-agent; comparators officialDatacurve board; reasoning max","family":"DeepSWE","locator":"PDF page4 performance table","metric":"percent","notes":"Opus74 omitted because Google current methodology explicitly identifies that board-rounded value as incorrect; underlying precision unresolved.","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":75.4,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"deepswe-1-1-percent","evidenceKind":"lab_self_report","harnessId":"deepswe-1-1-percent:meta-muse-1-3-report:MTEzdGFza3M7IE11c2UxLjNtaW5pLXN3ZS1hZ2VudDsgY29tcGFyYXRvcnMgb2ZmaWNpYWxEYXRhY3VydmUgYm9hcmQ7IHJlYXNvbmluZyB4aGlnaA","id":"launch-meta-muse-1-3-report-2028","ingestRunId":"launch-meta-muse-1-3-report","modelId":"muse-spark-1.2","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"113tasks; Muse1.3mini-swe-agent; comparators officialDatacurve board; reasoning xhigh","family":"DeepSWE","locator":"PDF page4 performance table","metric":"percent","notes":"Opus74 omitted because Google current methodology explicitly identifies that board-rounded value as incorrect; underlying precision unresolved.","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":55,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"deepswe-1-1-percent","evidenceKind":"lab_self_report","harnessId":"deepswe-1-1-percent:meta-muse-1-3-report:MTEzdGFza3M7IE11c2UxLjNtaW5pLXN3ZS1hZ2VudDsgY29tcGFyYXRvcnMgb2ZmaWNpYWxEYXRhY3VydmUgYm9hcmQ7IHJlYXNvbmluZyBtYXg","id":"launch-meta-muse-1-3-report-2029","ingestRunId":"launch-meta-muse-1-3-report","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"113tasks; Muse1.3mini-swe-agent; comparators officialDatacurve board; reasoning max","family":"DeepSWE","locator":"PDF page4 performance table","metric":"percent","notes":"Opus74 omitted because Google current methodology explicitly identifies that board-rounded value as incorrect; underlying precision unresolved.","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":73,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"swe-atlas-codebase-qna","evidenceKind":"lab_self_report","harnessId":"swe-atlas-codebase-qna:meta-muse-1-3-report:MTI0dGFza3MvMTFyZXBvczsgcHVibGljUW5BIG1pbmktc3dlLWFnZW50IHJ1YnJpYyBwYXNzQDE7IEdQVC9PcHVzIG93bmNhcmRzOyByZWFzb25pbmcgbWF4","id":"launch-meta-muse-1-3-report-2030","ingestRunId":"launch-meta-muse-1-3-report","modelId":"muse-spark-1.3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"124tasks/11repos; publicQnA mini-swe-agent rubric pass@1; GPT/Opus owncards; reasoning max","family":"SWE-Atlas Codebase QnA","locator":"PDF page4 performance table","metric":"percent","notes":"","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":59.4,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"swe-atlas-codebase-qna","evidenceKind":"lab_self_report","harnessId":"swe-atlas-codebase-qna:meta-muse-1-3-report:MTI0dGFza3MvMTFyZXBvczsgcHVibGljUW5BIG1pbmktc3dlLWFnZW50IHJ1YnJpYyBwYXNzQDE7IEdQVC9PcHVzIG93bmNhcmRzOyByZWFzb25pbmcgeGhpZ2g","id":"launch-meta-muse-1-3-report-2031","ingestRunId":"launch-meta-muse-1-3-report","modelId":"muse-spark-1.3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"124tasks/11repos; publicQnA mini-swe-agent rubric pass@1; GPT/Opus owncards; reasoning xhigh","family":"SWE-Atlas Codebase QnA","locator":"PDF page4 performance table","metric":"percent","notes":"","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":54,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"swe-atlas-codebase-qna","evidenceKind":"lab_self_report","harnessId":"swe-atlas-codebase-qna:meta-muse-1-3-report:MTI0dGFza3MvMTFyZXBvczsgcHVibGljUW5BIG1pbmktc3dlLWFnZW50IHJ1YnJpYyBwYXNzQDE7IEdQVC9PcHVzIG93bmNhcmRzOyByZWFzb25pbmcgeGhpZ2g","id":"launch-meta-muse-1-3-report-2032","ingestRunId":"launch-meta-muse-1-3-report","modelId":"muse-spark-1.2","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"124tasks/11repos; publicQnA mini-swe-agent rubric pass@1; GPT/Opus owncards; reasoning xhigh","family":"SWE-Atlas Codebase QnA","locator":"PDF page4 performance table","metric":"percent","notes":"","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":46.2,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"swe-atlas-codebase-qna","evidenceKind":"lab_self_report","harnessId":"swe-atlas-codebase-qna:meta-muse-1-3-report:MTI0dGFza3MvMTFyZXBvczsgcHVibGljUW5BIG1pbmktc3dlLWFnZW50IHJ1YnJpYyBwYXNzQDE7IEdQVC9PcHVzIG93bmNhcmRzOyByZWFzb25pbmcgbWF4","id":"launch-meta-muse-1-3-report-2033","ingestRunId":"launch-meta-muse-1-3-report","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"124tasks/11repos; publicQnA mini-swe-agent rubric pass@1; GPT/Opus owncards; reasoning max","family":"SWE-Atlas Codebase QnA","locator":"PDF page4 performance table","metric":"percent","notes":"","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":53.5,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"swe-atlas-codebase-qna","evidenceKind":"lab_self_report","harnessId":"swe-atlas-codebase-qna:meta-muse-1-3-report:MTI0dGFza3MvMTFyZXBvczsgcHVibGljUW5BIG1pbmktc3dlLWFnZW50IHJ1YnJpYyBwYXNzQDE7IEdQVC9PcHVzIG93bmNhcmRzOyByZWFzb25pbmcgbWF4","id":"launch-meta-muse-1-3-report-2034","ingestRunId":"launch-meta-muse-1-3-report","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"124tasks/11repos; publicQnA mini-swe-agent rubric pass@1; GPT/Opus owncards; reasoning max","family":"SWE-Atlas Codebase QnA","locator":"PDF page4 performance table","metric":"percent","notes":"","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":52.7,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"terminal-bench-2-1-percent-higher","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1-percent-higher:meta-muse-1-3-report:ODl0YXNrczsgZWFjaCBtb2RlbCBuYXRpdmVjb2RpbmcgaGFybmVzcyBpbiBpbnRlcm5hbGZyYW1ld29yay9jbG91ZHNhbmRib3g7IG9mZmljaWFsdmVyaWZpZXI7IEdQVCBvd25jYXJkOyByZWFzb25pbmcgbWF4OyBuYXRpdmUgaGFybmVzcyBmYW1pbHk6IE1ldGEgKGV4YWN0IGhhcm5lc3MgcmV2aXNpb24gbm90IHNwZWNpZmllZCk","id":"launch-meta-muse-1-3-report-2035","ingestRunId":"launch-meta-muse-1-3-report","modelId":"muse-spark-1.3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"89tasks; each model nativecoding harness in internalframework/cloudsandbox; officialverifier; GPT owncard; reasoning max; native harness family: Meta (exact harness revision not specified)","family":"Terminal-Bench","locator":"PDF page4 performance table","metric":"percent","notes":"Native harnesses differ; not a Terminus2-only comparison.","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":88.8,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"terminal-bench-2-1-percent-higher","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1-percent-higher:meta-muse-1-3-report:ODl0YXNrczsgZWFjaCBtb2RlbCBuYXRpdmVjb2RpbmcgaGFybmVzcyBpbiBpbnRlcm5hbGZyYW1ld29yay9jbG91ZHNhbmRib3g7IG9mZmljaWFsdmVyaWZpZXI7IEdQVCBvd25jYXJkOyByZWFzb25pbmcgeGhpZ2g7IG5hdGl2ZSBoYXJuZXNzIGZhbWlseTogTWV0YSAoZXhhY3QgaGFybmVzcyByZXZpc2lvbiBub3Qgc3BlY2lmaWVkKQ","id":"launch-meta-muse-1-3-report-2036","ingestRunId":"launch-meta-muse-1-3-report","modelId":"muse-spark-1.3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"89tasks; each model nativecoding harness in internalframework/cloudsandbox; officialverifier; GPT owncard; reasoning xhigh; native harness family: Meta (exact harness revision not specified)","family":"Terminal-Bench","locator":"PDF page4 performance table","metric":"percent","notes":"Native harnesses differ; not a Terminus2-only comparison.","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":89.2,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"terminal-bench-2-1-percent-higher","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1-percent-higher:meta-muse-1-3-report:ODl0YXNrczsgZWFjaCBtb2RlbCBuYXRpdmVjb2RpbmcgaGFybmVzcyBpbiBpbnRlcm5hbGZyYW1ld29yay9jbG91ZHNhbmRib3g7IG9mZmljaWFsdmVyaWZpZXI7IEdQVCBvd25jYXJkOyByZWFzb25pbmcgeGhpZ2g7IG5hdGl2ZSBoYXJuZXNzIGZhbWlseTogTWV0YSAoZXhhY3QgaGFybmVzcyByZXZpc2lvbiBub3Qgc3BlY2lmaWVkKQ","id":"launch-meta-muse-1-3-report-2037","ingestRunId":"launch-meta-muse-1-3-report","modelId":"muse-spark-1.2","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"89tasks; each model nativecoding harness in internalframework/cloudsandbox; officialverifier; GPT owncard; reasoning xhigh; native harness family: Meta (exact harness revision not specified)","family":"Terminal-Bench","locator":"PDF page4 performance table","metric":"percent","notes":"Native harnesses differ; not a Terminus2-only comparison.","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":82.9,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"terminal-bench-2-1-percent-higher","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1-percent-higher:meta-muse-1-3-report:ODl0YXNrczsgZWFjaCBtb2RlbCBuYXRpdmVjb2RpbmcgaGFybmVzcyBpbiBpbnRlcm5hbGZyYW1ld29yay9jbG91ZHNhbmRib3g7IG9mZmljaWFsdmVyaWZpZXI7IEdQVCBvd25jYXJkOyByZWFzb25pbmcgbWF4OyBuYXRpdmUgaGFybmVzcyBmYW1pbHk6IE9wZW5BSSAoZXhhY3QgaGFybmVzcyByZXZpc2lvbiBub3Qgc3BlY2lmaWVkKQ","id":"launch-meta-muse-1-3-report-2038","ingestRunId":"launch-meta-muse-1-3-report","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"89tasks; each model nativecoding harness in internalframework/cloudsandbox; officialverifier; GPT owncard; reasoning max; native harness family: OpenAI (exact harness revision not specified)","family":"Terminal-Bench","locator":"PDF page4 performance table","metric":"percent","notes":"Native harnesses differ; not a Terminus2-only comparison.","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":88.8,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"terminal-bench-2-1-percent-higher","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1-percent-higher:meta-muse-1-3-report:ODl0YXNrczsgZWFjaCBtb2RlbCBuYXRpdmVjb2RpbmcgaGFybmVzcyBpbiBpbnRlcm5hbGZyYW1ld29yay9jbG91ZHNhbmRib3g7IG9mZmljaWFsdmVyaWZpZXI7IEdQVCBvd25jYXJkOyByZWFzb25pbmcgbWF4OyBuYXRpdmUgaGFybmVzcyBmYW1pbHk6IEFudGhyb3BpYyAoZXhhY3QgaGFybmVzcyByZXZpc2lvbiBub3Qgc3BlY2lmaWVkKQ","id":"launch-meta-muse-1-3-report-2039","ingestRunId":"launch-meta-muse-1-3-report","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"89tasks; each model nativecoding harness in internalframework/cloudsandbox; officialverifier; GPT owncard; reasoning max; native harness family: Anthropic (exact harness revision not specified)","family":"Terminal-Bench","locator":"PDF page4 performance table","metric":"percent","notes":"Native harnesses differ; not a Terminus2-only comparison.","sourceId":"meta-muse-1-3-report","sourceTitle":"Muse Spark1.3 evaluation methodology"},"score":86.7,"scoreUnit":"percent","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology"},{"benchmarkId":"aime-2025-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"aime-2025-source-release-snapshot-version-not-specified:microsoft:TUFJIGF2ZyA0IHJ1bnMsIHRlbXBlcmF0dXJlIDEsIHRvcC1wIC45Nzsgc2ltcGxlIFJlQWN0IGJhc2gvc3RyaW5nLXJlcGxhY2U7IFRlcm1pbmFsLUJlbmNoIGlnbm9yZXMgcHJlZGVmaW5lZCB0aW1lb3V0cy4gQ29tcGFyYXRvciBjb25maWd1cmF0aW9ucyBmcm9tIGNpdGVkIHJlcG9ydHMu","id":"launch-microsoft-1375","ingestRunId":"launch-microsoft","modelId":"mai-thinking-1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports.","family":"AIME","locator":"Table 11, PDF page 53","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"microsoft","sourceTitle":"MAI-Thinking-1 technical report"},"score":97,"scoreUnit":"percent","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf"},{"benchmarkId":"aime-2025-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"aime-2025-source-release-snapshot-version-not-specified:microsoft:TUFJIGF2ZyA0IHJ1bnMsIHRlbXBlcmF0dXJlIDEsIHRvcC1wIC45Nzsgc2ltcGxlIFJlQWN0IGJhc2gvc3RyaW5nLXJlcGxhY2U7IFRlcm1pbmFsLUJlbmNoIGlnbm9yZXMgcHJlZGVmaW5lZCB0aW1lb3V0cy4gQ29tcGFyYXRvciBjb25maWd1cmF0aW9ucyBmcm9tIGNpdGVkIHJlcG9ydHMu","id":"launch-microsoft-1376","ingestRunId":"launch-microsoft","modelId":"claude-sonnet-4-6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports.","family":"AIME","locator":"Table 11, PDF page 53","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"microsoft","sourceTitle":"MAI-Thinking-1 technical report"},"score":95.6,"scoreUnit":"percent","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf"},{"benchmarkId":"aime-2025-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"aime-2025-source-release-snapshot-version-not-specified:microsoft:TUFJIGF2ZyA0IHJ1bnMsIHRlbXBlcmF0dXJlIDEsIHRvcC1wIC45Nzsgc2ltcGxlIFJlQWN0IGJhc2gvc3RyaW5nLXJlcGxhY2U7IFRlcm1pbmFsLUJlbmNoIGlnbm9yZXMgcHJlZGVmaW5lZCB0aW1lb3V0cy4gQ29tcGFyYXRvciBjb25maWd1cmF0aW9ucyBmcm9tIGNpdGVkIHJlcG9ydHMu","id":"launch-microsoft-1377","ingestRunId":"launch-microsoft","modelId":"claude-opus-4-6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports.","family":"AIME","locator":"Table 11, PDF page 53","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"microsoft","sourceTitle":"MAI-Thinking-1 technical report"},"score":99.8,"scoreUnit":"percent","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf"},{"benchmarkId":"aime-2026-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"aime-2026-source-release-snapshot-version-not-specified:microsoft:TUFJIGF2ZyA0IHJ1bnMsIHRlbXBlcmF0dXJlIDEsIHRvcC1wIC45Nzsgc2ltcGxlIFJlQWN0IGJhc2gvc3RyaW5nLXJlcGxhY2U7IFRlcm1pbmFsLUJlbmNoIGlnbm9yZXMgcHJlZGVmaW5lZCB0aW1lb3V0cy4gQ29tcGFyYXRvciBjb25maWd1cmF0aW9ucyBmcm9tIGNpdGVkIHJlcG9ydHMu","id":"launch-microsoft-1378","ingestRunId":"launch-microsoft","modelId":"mai-thinking-1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports.","family":"AIME","locator":"Table 11, PDF page 53","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"microsoft","sourceTitle":"MAI-Thinking-1 technical report"},"score":94.5,"scoreUnit":"percent","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf"},{"benchmarkId":"aime-2026-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"aime-2026-source-release-snapshot-version-not-specified:microsoft:TUFJIGF2ZyA0IHJ1bnMsIHRlbXBlcmF0dXJlIDEsIHRvcC1wIC45Nzsgc2ltcGxlIFJlQWN0IGJhc2gvc3RyaW5nLXJlcGxhY2U7IFRlcm1pbmFsLUJlbmNoIGlnbm9yZXMgcHJlZGVmaW5lZCB0aW1lb3V0cy4gQ29tcGFyYXRvciBjb25maWd1cmF0aW9ucyBmcm9tIGNpdGVkIHJlcG9ydHMu","id":"launch-microsoft-1379","ingestRunId":"launch-microsoft","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports.","family":"AIME","locator":"Table 11, PDF page 53","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"microsoft","sourceTitle":"MAI-Thinking-1 technical report"},"score":96.4,"scoreUnit":"percent","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf"},{"benchmarkId":"aime-2026-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"aime-2026-source-release-snapshot-version-not-specified:microsoft:TUFJIGF2ZyA0IHJ1bnMsIHRlbXBlcmF0dXJlIDEsIHRvcC1wIC45Nzsgc2ltcGxlIFJlQWN0IGJhc2gvc3RyaW5nLXJlcGxhY2U7IFRlcm1pbmFsLUJlbmNoIGlnbm9yZXMgcHJlZGVmaW5lZCB0aW1lb3V0cy4gQ29tcGFyYXRvciBjb25maWd1cmF0aW9ucyBmcm9tIGNpdGVkIHJlcG9ydHMu","id":"launch-microsoft-1380","ingestRunId":"launch-microsoft","modelId":"glm-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports.","family":"AIME","locator":"Table 11, PDF page 53","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"microsoft","sourceTitle":"MAI-Thinking-1 technical report"},"score":95.3,"scoreUnit":"percent","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf"},{"benchmarkId":"hmmt-february-2026-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hmmt-february-2026-source-release-snapshot-version-not-specified:microsoft:TUFJIGF2ZyA0IHJ1bnMsIHRlbXBlcmF0dXJlIDEsIHRvcC1wIC45Nzsgc2ltcGxlIFJlQWN0IGJhc2gvc3RyaW5nLXJlcGxhY2U7IFRlcm1pbmFsLUJlbmNoIGlnbm9yZXMgcHJlZGVmaW5lZCB0aW1lb3V0cy4gQ29tcGFyYXRvciBjb25maWd1cmF0aW9ucyBmcm9tIGNpdGVkIHJlcG9ydHMu","id":"launch-microsoft-1381","ingestRunId":"launch-microsoft","modelId":"mai-thinking-1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports.","family":"HMMT February 2026","locator":"Table 11, PDF page 53","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"microsoft","sourceTitle":"MAI-Thinking-1 technical report"},"score":84.9,"scoreUnit":"percent","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf"},{"benchmarkId":"hmmt-february-2026-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hmmt-february-2026-source-release-snapshot-version-not-specified:microsoft:TUFJIGF2ZyA0IHJ1bnMsIHRlbXBlcmF0dXJlIDEsIHRvcC1wIC45Nzsgc2ltcGxlIFJlQWN0IGJhc2gvc3RyaW5nLXJlcGxhY2U7IFRlcm1pbmFsLUJlbmNoIGlnbm9yZXMgcHJlZGVmaW5lZCB0aW1lb3V0cy4gQ29tcGFyYXRvciBjb25maWd1cmF0aW9ucyBmcm9tIGNpdGVkIHJlcG9ydHMu","id":"launch-microsoft-1382","ingestRunId":"launch-microsoft","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports.","family":"HMMT February 2026","locator":"Table 11, PDF page 53","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"microsoft","sourceTitle":"MAI-Thinking-1 technical report"},"score":92.7,"scoreUnit":"percent","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf"},{"benchmarkId":"hmmt-february-2026-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hmmt-february-2026-source-release-snapshot-version-not-specified:microsoft:TUFJIGF2ZyA0IHJ1bnMsIHRlbXBlcmF0dXJlIDEsIHRvcC1wIC45Nzsgc2ltcGxlIFJlQWN0IGJhc2gvc3RyaW5nLXJlcGxhY2U7IFRlcm1pbmFsLUJlbmNoIGlnbm9yZXMgcHJlZGVmaW5lZCB0aW1lb3V0cy4gQ29tcGFyYXRvciBjb25maWd1cmF0aW9ucyBmcm9tIGNpdGVkIHJlcG9ydHMu","id":"launch-microsoft-1383","ingestRunId":"launch-microsoft","modelId":"glm-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports.","family":"HMMT February 2026","locator":"Table 11, PDF page 53","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"microsoft","sourceTitle":"MAI-Thinking-1 technical report"},"score":82.6,"scoreUnit":"percent","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf"},{"benchmarkId":"gpqa-diamond-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gpqa-diamond-source-release-snapshot-version-not-specified:microsoft:TUFJIGF2ZyA0IHJ1bnMsIHRlbXBlcmF0dXJlIDEsIHRvcC1wIC45Nzsgc2ltcGxlIFJlQWN0IGJhc2gvc3RyaW5nLXJlcGxhY2U7IFRlcm1pbmFsLUJlbmNoIGlnbm9yZXMgcHJlZGVmaW5lZCB0aW1lb3V0cy4gQ29tcGFyYXRvciBjb25maWd1cmF0aW9ucyBmcm9tIGNpdGVkIHJlcG9ydHMu","id":"launch-microsoft-1384","ingestRunId":"launch-microsoft","modelId":"mai-thinking-1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports.","family":"GPQA Diamond","locator":"Table 11, PDF page 53","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"microsoft","sourceTitle":"MAI-Thinking-1 technical report"},"score":84.2,"scoreUnit":"percent","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf"},{"benchmarkId":"gpqa-diamond-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gpqa-diamond-source-release-snapshot-version-not-specified:microsoft:TUFJIGF2ZyA0IHJ1bnMsIHRlbXBlcmF0dXJlIDEsIHRvcC1wIC45Nzsgc2ltcGxlIFJlQWN0IGJhc2gvc3RyaW5nLXJlcGxhY2U7IFRlcm1pbmFsLUJlbmNoIGlnbm9yZXMgcHJlZGVmaW5lZCB0aW1lb3V0cy4gQ29tcGFyYXRvciBjb25maWd1cmF0aW9ucyBmcm9tIGNpdGVkIHJlcG9ydHMu","id":"launch-microsoft-1385","ingestRunId":"launch-microsoft","modelId":"claude-sonnet-4-6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports.","family":"GPQA Diamond","locator":"Table 11, PDF page 53","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"microsoft","sourceTitle":"MAI-Thinking-1 technical report"},"score":89.9,"scoreUnit":"percent","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf"},{"benchmarkId":"gpqa-diamond-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gpqa-diamond-source-release-snapshot-version-not-specified:microsoft:TUFJIGF2ZyA0IHJ1bnMsIHRlbXBlcmF0dXJlIDEsIHRvcC1wIC45Nzsgc2ltcGxlIFJlQWN0IGJhc2gvc3RyaW5nLXJlcGxhY2U7IFRlcm1pbmFsLUJlbmNoIGlnbm9yZXMgcHJlZGVmaW5lZCB0aW1lb3V0cy4gQ29tcGFyYXRvciBjb25maWd1cmF0aW9ucyBmcm9tIGNpdGVkIHJlcG9ydHMu","id":"launch-microsoft-1386","ingestRunId":"launch-microsoft","modelId":"claude-opus-4-6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports.","family":"GPQA Diamond","locator":"Table 11, PDF page 53","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"microsoft","sourceTitle":"MAI-Thinking-1 technical report"},"score":91.3,"scoreUnit":"percent","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf"},{"benchmarkId":"gpqa-diamond-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gpqa-diamond-source-release-snapshot-version-not-specified:microsoft:TUFJIGF2ZyA0IHJ1bnMsIHRlbXBlcmF0dXJlIDEsIHRvcC1wIC45Nzsgc2ltcGxlIFJlQWN0IGJhc2gvc3RyaW5nLXJlcGxhY2U7IFRlcm1pbmFsLUJlbmNoIGlnbm9yZXMgcHJlZGVmaW5lZCB0aW1lb3V0cy4gQ29tcGFyYXRvciBjb25maWd1cmF0aW9ucyBmcm9tIGNpdGVkIHJlcG9ydHMu","id":"launch-microsoft-1387","ingestRunId":"launch-microsoft","modelId":"gpt-5.4","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports.","family":"GPQA Diamond","locator":"Table 11, PDF page 53","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"microsoft","sourceTitle":"MAI-Thinking-1 technical report"},"score":92.8,"scoreUnit":"percent","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf"},{"benchmarkId":"gpqa-diamond-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gpqa-diamond-source-release-snapshot-version-not-specified:microsoft:TUFJIGF2ZyA0IHJ1bnMsIHRlbXBlcmF0dXJlIDEsIHRvcC1wIC45Nzsgc2ltcGxlIFJlQWN0IGJhc2gvc3RyaW5nLXJlcGxhY2U7IFRlcm1pbmFsLUJlbmNoIGlnbm9yZXMgcHJlZGVmaW5lZCB0aW1lb3V0cy4gQ29tcGFyYXRvciBjb25maWd1cmF0aW9ucyBmcm9tIGNpdGVkIHJlcG9ydHMu","id":"launch-microsoft-1388","ingestRunId":"launch-microsoft","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports.","family":"GPQA Diamond","locator":"Table 11, PDF page 53","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"microsoft","sourceTitle":"MAI-Thinking-1 technical report"},"score":90.5,"scoreUnit":"percent","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf"},{"benchmarkId":"gpqa-diamond-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gpqa-diamond-source-release-snapshot-version-not-specified:microsoft:TUFJIGF2ZyA0IHJ1bnMsIHRlbXBlcmF0dXJlIDEsIHRvcC1wIC45Nzsgc2ltcGxlIFJlQWN0IGJhc2gvc3RyaW5nLXJlcGxhY2U7IFRlcm1pbmFsLUJlbmNoIGlnbm9yZXMgcHJlZGVmaW5lZCB0aW1lb3V0cy4gQ29tcGFyYXRvciBjb25maWd1cmF0aW9ucyBmcm9tIGNpdGVkIHJlcG9ydHMu","id":"launch-microsoft-1389","ingestRunId":"launch-microsoft","modelId":"glm-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports.","family":"GPQA Diamond","locator":"Table 11, PDF page 53","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"microsoft","sourceTitle":"MAI-Thinking-1 technical report"},"score":86.2,"scoreUnit":"percent","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf"},{"benchmarkId":"livecodebench-v6-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"livecodebench-v6-source-release-snapshot-version-not-specified-pct:microsoft:TUFJIGF2ZyA0IHJ1bnMsIHRlbXBlcmF0dXJlIDEsIHRvcC1wIC45Nzsgc2ltcGxlIFJlQWN0IGJhc2gvc3RyaW5nLXJlcGxhY2U7IFRlcm1pbmFsLUJlbmNoIGlnbm9yZXMgcHJlZGVmaW5lZCB0aW1lb3V0cy4gQ29tcGFyYXRvciBjb25maWd1cmF0aW9ucyBmcm9tIGNpdGVkIHJlcG9ydHMu","id":"launch-microsoft-1390","ingestRunId":"launch-microsoft","modelId":"mai-thinking-1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports.","family":"LiveCodeBench","locator":"Table 11, PDF page 53","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"microsoft","sourceTitle":"MAI-Thinking-1 technical report"},"score":87.7,"scoreUnit":"percent","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf"},{"benchmarkId":"livecodebench-v6-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"livecodebench-v6-source-release-snapshot-version-not-specified-pct:microsoft:TUFJIGF2ZyA0IHJ1bnMsIHRlbXBlcmF0dXJlIDEsIHRvcC1wIC45Nzsgc2ltcGxlIFJlQWN0IGJhc2gvc3RyaW5nLXJlcGxhY2U7IFRlcm1pbmFsLUJlbmNoIGlnbm9yZXMgcHJlZGVmaW5lZCB0aW1lb3V0cy4gQ29tcGFyYXRvciBjb25maWd1cmF0aW9ucyBmcm9tIGNpdGVkIHJlcG9ydHMu","id":"launch-microsoft-1391","ingestRunId":"launch-microsoft","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports.","family":"LiveCodeBench","locator":"Table 11, PDF page 53","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"microsoft","sourceTitle":"MAI-Thinking-1 technical report"},"score":89.6,"scoreUnit":"percent","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf"},{"benchmarkId":"terminal-bench-2-0","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-0:microsoft:TUFJIGF2ZyA0IHJ1bnMsIHRlbXBlcmF0dXJlIDEsIHRvcC1wIC45Nzsgc2ltcGxlIFJlQWN0IGJhc2gvc3RyaW5nLXJlcGxhY2U7IFRlcm1pbmFsLUJlbmNoIGlnbm9yZXMgcHJlZGVmaW5lZCB0aW1lb3V0cy4gQ29tcGFyYXRvciBjb25maWd1cmF0aW9ucyBmcm9tIGNpdGVkIHJlcG9ydHMu","id":"launch-microsoft-1392","ingestRunId":"launch-microsoft","modelId":"mai-thinking-1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports.","family":"Terminal-Bench","locator":"Table 11, PDF page 53","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"microsoft","sourceTitle":"MAI-Thinking-1 technical report"},"score":46,"scoreUnit":"percent","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf"},{"benchmarkId":"terminal-bench-2-0","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-0:microsoft:TUFJIGF2ZyA0IHJ1bnMsIHRlbXBlcmF0dXJlIDEsIHRvcC1wIC45Nzsgc2ltcGxlIFJlQWN0IGJhc2gvc3RyaW5nLXJlcGxhY2U7IFRlcm1pbmFsLUJlbmNoIGlnbm9yZXMgcHJlZGVmaW5lZCB0aW1lb3V0cy4gQ29tcGFyYXRvciBjb25maWd1cmF0aW9ucyBmcm9tIGNpdGVkIHJlcG9ydHMu","id":"launch-microsoft-1393","ingestRunId":"launch-microsoft","modelId":"claude-sonnet-4-6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports.","family":"Terminal-Bench","locator":"Table 11, PDF page 53","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"microsoft","sourceTitle":"MAI-Thinking-1 technical report"},"score":59.1,"scoreUnit":"percent","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf"},{"benchmarkId":"terminal-bench-2-0","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-0:microsoft:TUFJIGF2ZyA0IHJ1bnMsIHRlbXBlcmF0dXJlIDEsIHRvcC1wIC45Nzsgc2ltcGxlIFJlQWN0IGJhc2gvc3RyaW5nLXJlcGxhY2U7IFRlcm1pbmFsLUJlbmNoIGlnbm9yZXMgcHJlZGVmaW5lZCB0aW1lb3V0cy4gQ29tcGFyYXRvciBjb25maWd1cmF0aW9ucyBmcm9tIGNpdGVkIHJlcG9ydHMu","id":"launch-microsoft-1394","ingestRunId":"launch-microsoft","modelId":"claude-opus-4-6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports.","family":"Terminal-Bench","locator":"Table 11, PDF page 53","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"microsoft","sourceTitle":"MAI-Thinking-1 technical report"},"score":65.4,"scoreUnit":"percent","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf"},{"benchmarkId":"terminal-bench-2-0","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-0:microsoft:TUFJIGF2ZyA0IHJ1bnMsIHRlbXBlcmF0dXJlIDEsIHRvcC1wIC45Nzsgc2ltcGxlIFJlQWN0IGJhc2gvc3RyaW5nLXJlcGxhY2U7IFRlcm1pbmFsLUJlbmNoIGlnbm9yZXMgcHJlZGVmaW5lZCB0aW1lb3V0cy4gQ29tcGFyYXRvciBjb25maWd1cmF0aW9ucyBmcm9tIGNpdGVkIHJlcG9ydHMu","id":"launch-microsoft-1395","ingestRunId":"launch-microsoft","modelId":"gpt-5.4","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports.","family":"Terminal-Bench","locator":"Table 11, PDF page 53","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"microsoft","sourceTitle":"MAI-Thinking-1 technical report"},"score":75.1,"scoreUnit":"percent","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf"},{"benchmarkId":"terminal-bench-2-0","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-0:microsoft:TUFJIGF2ZyA0IHJ1bnMsIHRlbXBlcmF0dXJlIDEsIHRvcC1wIC45Nzsgc2ltcGxlIFJlQWN0IGJhc2gvc3RyaW5nLXJlcGxhY2U7IFRlcm1pbmFsLUJlbmNoIGlnbm9yZXMgcHJlZGVmaW5lZCB0aW1lb3V0cy4gQ29tcGFyYXRvciBjb25maWd1cmF0aW9ucyBmcm9tIGNpdGVkIHJlcG9ydHMu","id":"launch-microsoft-1396","ingestRunId":"launch-microsoft","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports.","family":"Terminal-Bench","locator":"Table 11, PDF page 53","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"microsoft","sourceTitle":"MAI-Thinking-1 technical report"},"score":66.7,"scoreUnit":"percent","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf"},{"benchmarkId":"terminal-bench-2-0","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-0:microsoft:TUFJIGF2ZyA0IHJ1bnMsIHRlbXBlcmF0dXJlIDEsIHRvcC1wIC45Nzsgc2ltcGxlIFJlQWN0IGJhc2gvc3RyaW5nLXJlcGxhY2U7IFRlcm1pbmFsLUJlbmNoIGlnbm9yZXMgcHJlZGVmaW5lZCB0aW1lb3V0cy4gQ29tcGFyYXRvciBjb25maWd1cmF0aW9ucyBmcm9tIGNpdGVkIHJlcG9ydHMu","id":"launch-microsoft-1397","ingestRunId":"launch-microsoft","modelId":"glm-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports.","family":"Terminal-Bench","locator":"Table 11, PDF page 53","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"microsoft","sourceTitle":"MAI-Thinking-1 technical report"},"score":69,"scoreUnit":"percent","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf"},{"benchmarkId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct:microsoft:TUFJIGF2ZyA0IHJ1bnMsIHRlbXBlcmF0dXJlIDEsIHRvcC1wIC45Nzsgc2ltcGxlIFJlQWN0IGJhc2gvc3RyaW5nLXJlcGxhY2U7IFRlcm1pbmFsLUJlbmNoIGlnbm9yZXMgcHJlZGVmaW5lZCB0aW1lb3V0cy4gQ29tcGFyYXRvciBjb25maWd1cmF0aW9ucyBmcm9tIGNpdGVkIHJlcG9ydHMu","id":"launch-microsoft-1398","ingestRunId":"launch-microsoft","modelId":"mai-thinking-1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports.","family":"SWE-bench Verified","locator":"Table 11, PDF page 53","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"microsoft","sourceTitle":"MAI-Thinking-1 technical report"},"score":73.5,"scoreUnit":"percent","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf"},{"benchmarkId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct:microsoft:TUFJIGF2ZyA0IHJ1bnMsIHRlbXBlcmF0dXJlIDEsIHRvcC1wIC45Nzsgc2ltcGxlIFJlQWN0IGJhc2gvc3RyaW5nLXJlcGxhY2U7IFRlcm1pbmFsLUJlbmNoIGlnbm9yZXMgcHJlZGVmaW5lZCB0aW1lb3V0cy4gQ29tcGFyYXRvciBjb25maWd1cmF0aW9ucyBmcm9tIGNpdGVkIHJlcG9ydHMu","id":"launch-microsoft-1399","ingestRunId":"launch-microsoft","modelId":"claude-sonnet-4-6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports.","family":"SWE-bench Verified","locator":"Table 11, PDF page 53","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"microsoft","sourceTitle":"MAI-Thinking-1 technical report"},"score":79.6,"scoreUnit":"percent","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf"},{"benchmarkId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct:microsoft:TUFJIGF2ZyA0IHJ1bnMsIHRlbXBlcmF0dXJlIDEsIHRvcC1wIC45Nzsgc2ltcGxlIFJlQWN0IGJhc2gvc3RyaW5nLXJlcGxhY2U7IFRlcm1pbmFsLUJlbmNoIGlnbm9yZXMgcHJlZGVmaW5lZCB0aW1lb3V0cy4gQ29tcGFyYXRvciBjb25maWd1cmF0aW9ucyBmcm9tIGNpdGVkIHJlcG9ydHMu","id":"launch-microsoft-1400","ingestRunId":"launch-microsoft","modelId":"claude-opus-4-6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports.","family":"SWE-bench Verified","locator":"Table 11, PDF page 53","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"microsoft","sourceTitle":"MAI-Thinking-1 technical report"},"score":80.8,"scoreUnit":"percent","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf"},{"benchmarkId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct:microsoft:TUFJIGF2ZyA0IHJ1bnMsIHRlbXBlcmF0dXJlIDEsIHRvcC1wIC45Nzsgc2ltcGxlIFJlQWN0IGJhc2gvc3RyaW5nLXJlcGxhY2U7IFRlcm1pbmFsLUJlbmNoIGlnbm9yZXMgcHJlZGVmaW5lZCB0aW1lb3V0cy4gQ29tcGFyYXRvciBjb25maWd1cmF0aW9ucyBmcm9tIGNpdGVkIHJlcG9ydHMu","id":"launch-microsoft-1401","ingestRunId":"launch-microsoft","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports.","family":"SWE-bench Verified","locator":"Table 11, PDF page 53","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"microsoft","sourceTitle":"MAI-Thinking-1 technical report"},"score":80.2,"scoreUnit":"percent","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf"},{"benchmarkId":"swe-bench-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"swe-bench-pro-source-release-snapshot-version-not-specified:microsoft:TUFJIGF2ZyA0IHJ1bnMsIHRlbXBlcmF0dXJlIDEsIHRvcC1wIC45Nzsgc2ltcGxlIFJlQWN0IGJhc2gvc3RyaW5nLXJlcGxhY2U7IFRlcm1pbmFsLUJlbmNoIGlnbm9yZXMgcHJlZGVmaW5lZCB0aW1lb3V0cy4gQ29tcGFyYXRvciBjb25maWd1cmF0aW9ucyBmcm9tIGNpdGVkIHJlcG9ydHMu","id":"launch-microsoft-1402","ingestRunId":"launch-microsoft","modelId":"mai-thinking-1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports.","family":"SWE-bench Pro","locator":"Table 11, PDF page 53","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"microsoft","sourceTitle":"MAI-Thinking-1 technical report"},"score":52.8,"scoreUnit":"percent","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf"},{"benchmarkId":"swe-bench-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"swe-bench-pro-source-release-snapshot-version-not-specified:microsoft:TUFJIGF2ZyA0IHJ1bnMsIHRlbXBlcmF0dXJlIDEsIHRvcC1wIC45Nzsgc2ltcGxlIFJlQWN0IGJhc2gvc3RyaW5nLXJlcGxhY2U7IFRlcm1pbmFsLUJlbmNoIGlnbm9yZXMgcHJlZGVmaW5lZCB0aW1lb3V0cy4gQ29tcGFyYXRvciBjb25maWd1cmF0aW9ucyBmcm9tIGNpdGVkIHJlcG9ydHMu","id":"launch-microsoft-1403","ingestRunId":"launch-microsoft","modelId":"claude-opus-4-6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports.","family":"SWE-bench Pro","locator":"Table 11, PDF page 53","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"microsoft","sourceTitle":"MAI-Thinking-1 technical report"},"score":53.4,"scoreUnit":"percent","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf"},{"benchmarkId":"swe-bench-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"swe-bench-pro-source-release-snapshot-version-not-specified:microsoft:TUFJIGF2ZyA0IHJ1bnMsIHRlbXBlcmF0dXJlIDEsIHRvcC1wIC45Nzsgc2ltcGxlIFJlQWN0IGJhc2gvc3RyaW5nLXJlcGxhY2U7IFRlcm1pbmFsLUJlbmNoIGlnbm9yZXMgcHJlZGVmaW5lZCB0aW1lb3V0cy4gQ29tcGFyYXRvciBjb25maWd1cmF0aW9ucyBmcm9tIGNpdGVkIHJlcG9ydHMu","id":"launch-microsoft-1404","ingestRunId":"launch-microsoft","modelId":"gpt-5.4","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports.","family":"SWE-bench Pro","locator":"Table 11, PDF page 53","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"microsoft","sourceTitle":"MAI-Thinking-1 technical report"},"score":57.7,"scoreUnit":"percent","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf"},{"benchmarkId":"swe-bench-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"swe-bench-pro-source-release-snapshot-version-not-specified:microsoft:TUFJIGF2ZyA0IHJ1bnMsIHRlbXBlcmF0dXJlIDEsIHRvcC1wIC45Nzsgc2ltcGxlIFJlQWN0IGJhc2gvc3RyaW5nLXJlcGxhY2U7IFRlcm1pbmFsLUJlbmNoIGlnbm9yZXMgcHJlZGVmaW5lZCB0aW1lb3V0cy4gQ29tcGFyYXRvciBjb25maWd1cmF0aW9ucyBmcm9tIGNpdGVkIHJlcG9ydHMu","id":"launch-microsoft-1405","ingestRunId":"launch-microsoft","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports.","family":"SWE-bench Pro","locator":"Table 11, PDF page 53","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"microsoft","sourceTitle":"MAI-Thinking-1 technical report"},"score":58.6,"scoreUnit":"percent","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf"},{"benchmarkId":"swe-bench-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"swe-bench-pro-source-release-snapshot-version-not-specified:microsoft:TUFJIGF2ZyA0IHJ1bnMsIHRlbXBlcmF0dXJlIDEsIHRvcC1wIC45Nzsgc2ltcGxlIFJlQWN0IGJhc2gvc3RyaW5nLXJlcGxhY2U7IFRlcm1pbmFsLUJlbmNoIGlnbm9yZXMgcHJlZGVmaW5lZCB0aW1lb3V0cy4gQ29tcGFyYXRvciBjb25maWd1cmF0aW9ucyBmcm9tIGNpdGVkIHJlcG9ydHMu","id":"launch-microsoft-1406","ingestRunId":"launch-microsoft","modelId":"glm-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports.","family":"SWE-bench Pro","locator":"Table 11, PDF page 53","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"microsoft","sourceTitle":"MAI-Thinking-1 technical report"},"score":58.4,"scoreUnit":"percent","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf"},{"benchmarkId":"mmlu-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mmlu-pro-source-release-snapshot-version-not-specified:microsoft:TWljcm9zb2Z0IG93biBldmFsdWF0aW9uIHN1aXRlOyBTb25uZXQgbWF4IHJlYXNvbmluZzsgVGFibGVzIDEyLzE5LiBMb25nQmVuY2hWMiBjYXBwZWQgMjU2ayw0MDggcXVlc3Rpb25zOyBDb3JwdXNRQSBHUFQtNS40IGhpZ2gganVkZ2U7IDQgcnVucy4","id":"launch-microsoft-1407","ingestRunId":"launch-microsoft","modelId":"mai-thinking-1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs.","family":"MMLU-Pro","locator":"Tables 12 and 19","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"microsoft","sourceTitle":"MAI-Thinking-1 technical report"},"score":85,"scoreUnit":"percent","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf"},{"benchmarkId":"mmlu-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mmlu-pro-source-release-snapshot-version-not-specified:microsoft:TWljcm9zb2Z0IG93biBldmFsdWF0aW9uIHN1aXRlOyBTb25uZXQgbWF4IHJlYXNvbmluZzsgVGFibGVzIDEyLzE5LiBMb25nQmVuY2hWMiBjYXBwZWQgMjU2ayw0MDggcXVlc3Rpb25zOyBDb3JwdXNRQSBHUFQtNS40IGhpZ2gganVkZ2U7IDQgcnVucy4","id":"launch-microsoft-1408","ingestRunId":"launch-microsoft","modelId":"claude-sonnet-4-6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs.","family":"MMLU-Pro","locator":"Tables 12 and 19","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"microsoft","sourceTitle":"MAI-Thinking-1 technical report"},"score":87,"scoreUnit":"percent","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf"},{"benchmarkId":"simpleqa-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"simpleqa-verified-source-release-snapshot-version-not-specified:microsoft:TWljcm9zb2Z0IG93biBldmFsdWF0aW9uIHN1aXRlOyBTb25uZXQgbWF4IHJlYXNvbmluZzsgVGFibGVzIDEyLzE5LiBMb25nQmVuY2hWMiBjYXBwZWQgMjU2ayw0MDggcXVlc3Rpb25zOyBDb3JwdXNRQSBHUFQtNS40IGhpZ2gganVkZ2U7IDQgcnVucy4","id":"launch-microsoft-1409","ingestRunId":"launch-microsoft","modelId":"mai-thinking-1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs.","family":"SimpleQA Verified","locator":"Tables 12 and 19","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"microsoft","sourceTitle":"MAI-Thinking-1 technical report"},"score":31,"scoreUnit":"percent","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf"},{"benchmarkId":"simpleqa-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"simpleqa-verified-source-release-snapshot-version-not-specified:microsoft:TWljcm9zb2Z0IG93biBldmFsdWF0aW9uIHN1aXRlOyBTb25uZXQgbWF4IHJlYXNvbmluZzsgVGFibGVzIDEyLzE5LiBMb25nQmVuY2hWMiBjYXBwZWQgMjU2ayw0MDggcXVlc3Rpb25zOyBDb3JwdXNRQSBHUFQtNS40IGhpZ2gganVkZ2U7IDQgcnVucy4","id":"launch-microsoft-1410","ingestRunId":"launch-microsoft","modelId":"claude-sonnet-4-6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs.","family":"SimpleQA Verified","locator":"Tables 12 and 19","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"microsoft","sourceTitle":"MAI-Thinking-1 technical report"},"score":29,"scoreUnit":"percent","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf"},{"benchmarkId":"if-bench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"if-bench-source-release-snapshot-version-not-specified:microsoft:TWljcm9zb2Z0IG93biBldmFsdWF0aW9uIHN1aXRlOyBTb25uZXQgbWF4IHJlYXNvbmluZzsgVGFibGVzIDEyLzE5LiBMb25nQmVuY2hWMiBjYXBwZWQgMjU2ayw0MDggcXVlc3Rpb25zOyBDb3JwdXNRQSBHUFQtNS40IGhpZ2gganVkZ2U7IDQgcnVucy4","id":"launch-microsoft-1411","ingestRunId":"launch-microsoft","modelId":"mai-thinking-1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs.","family":"IF Bench","locator":"Tables 12 and 19","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"microsoft","sourceTitle":"MAI-Thinking-1 technical report"},"score":69,"scoreUnit":"percent","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf"},{"benchmarkId":"if-bench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"if-bench-source-release-snapshot-version-not-specified:microsoft:TWljcm9zb2Z0IG93biBldmFsdWF0aW9uIHN1aXRlOyBTb25uZXQgbWF4IHJlYXNvbmluZzsgVGFibGVzIDEyLzE5LiBMb25nQmVuY2hWMiBjYXBwZWQgMjU2ayw0MDggcXVlc3Rpb25zOyBDb3JwdXNRQSBHUFQtNS40IGhpZ2gganVkZ2U7IDQgcnVucy4","id":"launch-microsoft-1412","ingestRunId":"launch-microsoft","modelId":"claude-sonnet-4-6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs.","family":"IF Bench","locator":"Tables 12 and 19","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"microsoft","sourceTitle":"MAI-Thinking-1 technical report"},"score":50,"scoreUnit":"percent","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf"},{"benchmarkId":"advancedif-rubric-level-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"advancedif-rubric-level-source-release-snapshot-version-not-specified:microsoft:TWljcm9zb2Z0IG93biBldmFsdWF0aW9uIHN1aXRlOyBTb25uZXQgbWF4IHJlYXNvbmluZzsgVGFibGVzIDEyLzE5LiBMb25nQmVuY2hWMiBjYXBwZWQgMjU2ayw0MDggcXVlc3Rpb25zOyBDb3JwdXNRQSBHUFQtNS40IGhpZ2gganVkZ2U7IDQgcnVucy4","id":"launch-microsoft-1413","ingestRunId":"launch-microsoft","modelId":"mai-thinking-1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs.","family":"AdvancedIF rubric-level","locator":"Tables 12 and 19","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"microsoft","sourceTitle":"MAI-Thinking-1 technical report"},"score":85,"scoreUnit":"percent","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf"},{"benchmarkId":"advancedif-rubric-level-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"advancedif-rubric-level-source-release-snapshot-version-not-specified:microsoft:TWljcm9zb2Z0IG93biBldmFsdWF0aW9uIHN1aXRlOyBTb25uZXQgbWF4IHJlYXNvbmluZzsgVGFibGVzIDEyLzE5LiBMb25nQmVuY2hWMiBjYXBwZWQgMjU2ayw0MDggcXVlc3Rpb25zOyBDb3JwdXNRQSBHUFQtNS40IGhpZ2gganVkZ2U7IDQgcnVucy4","id":"launch-microsoft-1414","ingestRunId":"launch-microsoft","modelId":"claude-sonnet-4-6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs.","family":"AdvancedIF rubric-level","locator":"Tables 12 and 19","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"microsoft","sourceTitle":"MAI-Thinking-1 technical report"},"score":86,"scoreUnit":"percent","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf"},{"benchmarkId":"multichallenge-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"multichallenge-source-release-snapshot-version-not-specified:microsoft:TWljcm9zb2Z0IG93biBldmFsdWF0aW9uIHN1aXRlOyBTb25uZXQgbWF4IHJlYXNvbmluZzsgVGFibGVzIDEyLzE5LiBMb25nQmVuY2hWMiBjYXBwZWQgMjU2ayw0MDggcXVlc3Rpb25zOyBDb3JwdXNRQSBHUFQtNS40IGhpZ2gganVkZ2U7IDQgcnVucy4","id":"launch-microsoft-1415","ingestRunId":"launch-microsoft","modelId":"mai-thinking-1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs.","family":"MultiChallenge","locator":"Tables 12 and 19","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"microsoft","sourceTitle":"MAI-Thinking-1 technical report"},"score":53,"scoreUnit":"percent","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf"},{"benchmarkId":"multichallenge-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"multichallenge-source-release-snapshot-version-not-specified:microsoft:TWljcm9zb2Z0IG93biBldmFsdWF0aW9uIHN1aXRlOyBTb25uZXQgbWF4IHJlYXNvbmluZzsgVGFibGVzIDEyLzE5LiBMb25nQmVuY2hWMiBjYXBwZWQgMjU2ayw0MDggcXVlc3Rpb25zOyBDb3JwdXNRQSBHUFQtNS40IGhpZ2gganVkZ2U7IDQgcnVucy4","id":"launch-microsoft-1416","ingestRunId":"launch-microsoft","modelId":"claude-sonnet-4-6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs.","family":"MultiChallenge","locator":"Tables 12 and 19","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"microsoft","sourceTitle":"MAI-Thinking-1 technical report"},"score":57,"scoreUnit":"percent","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf"},{"benchmarkId":"graphwalks-128k-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"graphwalks-128k-source-release-snapshot-version-not-specified:microsoft:TWljcm9zb2Z0IG93biBldmFsdWF0aW9uIHN1aXRlOyBTb25uZXQgbWF4IHJlYXNvbmluZzsgVGFibGVzIDEyLzE5LiBMb25nQmVuY2hWMiBjYXBwZWQgMjU2ayw0MDggcXVlc3Rpb25zOyBDb3JwdXNRQSBHUFQtNS40IGhpZ2gganVkZ2U7IDQgcnVucy4","id":"launch-microsoft-1417","ingestRunId":"launch-microsoft","modelId":"mai-thinking-1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs.","family":"GraphWalks <=128k","locator":"Tables 12 and 19","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"microsoft","sourceTitle":"MAI-Thinking-1 technical report"},"score":90,"scoreUnit":"percent","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf"},{"benchmarkId":"graphwalks-128k-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"graphwalks-128k-source-release-snapshot-version-not-specified:microsoft:TWljcm9zb2Z0IG93biBldmFsdWF0aW9uIHN1aXRlOyBTb25uZXQgbWF4IHJlYXNvbmluZzsgVGFibGVzIDEyLzE5LiBMb25nQmVuY2hWMiBjYXBwZWQgMjU2ayw0MDggcXVlc3Rpb25zOyBDb3JwdXNRQSBHUFQtNS40IGhpZ2gganVkZ2U7IDQgcnVucy4","id":"launch-microsoft-1418","ingestRunId":"launch-microsoft","modelId":"claude-sonnet-4-6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs.","family":"GraphWalks <=128k","locator":"Tables 12 and 19","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"microsoft","sourceTitle":"MAI-Thinking-1 technical report"},"score":96,"scoreUnit":"percent","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf"},{"benchmarkId":"bfcl-v3-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"bfcl-v3-source-release-snapshot-version-not-specified:microsoft:TWljcm9zb2Z0IG93biBldmFsdWF0aW9uIHN1aXRlOyBTb25uZXQgbWF4IHJlYXNvbmluZzsgVGFibGVzIDEyLzE5LiBMb25nQmVuY2hWMiBjYXBwZWQgMjU2ayw0MDggcXVlc3Rpb25zOyBDb3JwdXNRQSBHUFQtNS40IGhpZ2gganVkZ2U7IDQgcnVucy4","id":"launch-microsoft-1419","ingestRunId":"launch-microsoft","modelId":"mai-thinking-1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs.","family":"BFCL v3","locator":"Tables 12 and 19","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"microsoft","sourceTitle":"MAI-Thinking-1 technical report"},"score":72,"scoreUnit":"percent","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf"},{"benchmarkId":"bfcl-v3-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"bfcl-v3-source-release-snapshot-version-not-specified:microsoft:TWljcm9zb2Z0IG93biBldmFsdWF0aW9uIHN1aXRlOyBTb25uZXQgbWF4IHJlYXNvbmluZzsgVGFibGVzIDEyLzE5LiBMb25nQmVuY2hWMiBjYXBwZWQgMjU2ayw0MDggcXVlc3Rpb25zOyBDb3JwdXNRQSBHUFQtNS40IGhpZ2gganVkZ2U7IDQgcnVucy4","id":"launch-microsoft-1420","ingestRunId":"launch-microsoft","modelId":"claude-sonnet-4-6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs.","family":"BFCL v3","locator":"Tables 12 and 19","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"microsoft","sourceTitle":"MAI-Thinking-1 technical report"},"score":76,"scoreUnit":"percent","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf"},{"benchmarkId":"healthbench-professional-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"healthbench-professional-source-release-snapshot-version-not-specified:microsoft:TWljcm9zb2Z0IG93biBldmFsdWF0aW9uIHN1aXRlOyBTb25uZXQgbWF4IHJlYXNvbmluZzsgVGFibGVzIDEyLzE5LiBMb25nQmVuY2hWMiBjYXBwZWQgMjU2ayw0MDggcXVlc3Rpb25zOyBDb3JwdXNRQSBHUFQtNS40IGhpZ2gganVkZ2U7IDQgcnVucy4","id":"launch-microsoft-1421","ingestRunId":"launch-microsoft","modelId":"mai-thinking-1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs.","family":"HealthBench Professional","locator":"Tables 12 and 19","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"microsoft","sourceTitle":"MAI-Thinking-1 technical report"},"score":35,"scoreUnit":"percent","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf"},{"benchmarkId":"healthbench-professional-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"healthbench-professional-source-release-snapshot-version-not-specified:microsoft:TWljcm9zb2Z0IG93biBldmFsdWF0aW9uIHN1aXRlOyBTb25uZXQgbWF4IHJlYXNvbmluZzsgVGFibGVzIDEyLzE5LiBMb25nQmVuY2hWMiBjYXBwZWQgMjU2ayw0MDggcXVlc3Rpb25zOyBDb3JwdXNRQSBHUFQtNS40IGhpZ2gganVkZ2U7IDQgcnVucy4","id":"launch-microsoft-1422","ingestRunId":"launch-microsoft","modelId":"claude-sonnet-4-6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs.","family":"HealthBench Professional","locator":"Tables 12 and 19","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"microsoft","sourceTitle":"MAI-Thinking-1 technical report"},"score":38,"scoreUnit":"percent","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf"},{"benchmarkId":"medxpertqa-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"medxpertqa-source-release-snapshot-version-not-specified:microsoft:TWljcm9zb2Z0IG93biBldmFsdWF0aW9uIHN1aXRlOyBTb25uZXQgbWF4IHJlYXNvbmluZzsgVGFibGVzIDEyLzE5LiBMb25nQmVuY2hWMiBjYXBwZWQgMjU2ayw0MDggcXVlc3Rpb25zOyBDb3JwdXNRQSBHUFQtNS40IGhpZ2gganVkZ2U7IDQgcnVucy4","id":"launch-microsoft-1423","ingestRunId":"launch-microsoft","modelId":"mai-thinking-1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs.","family":"MedXpertQA","locator":"Tables 12 and 19","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"microsoft","sourceTitle":"MAI-Thinking-1 technical report"},"score":43,"scoreUnit":"percent","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf"},{"benchmarkId":"medxpertqa-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"medxpertqa-source-release-snapshot-version-not-specified:microsoft:TWljcm9zb2Z0IG93biBldmFsdWF0aW9uIHN1aXRlOyBTb25uZXQgbWF4IHJlYXNvbmluZzsgVGFibGVzIDEyLzE5LiBMb25nQmVuY2hWMiBjYXBwZWQgMjU2ayw0MDggcXVlc3Rpb25zOyBDb3JwdXNRQSBHUFQtNS40IGhpZ2gganVkZ2U7IDQgcnVucy4","id":"launch-microsoft-1424","ingestRunId":"launch-microsoft","modelId":"claude-sonnet-4-6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs.","family":"MedXpertQA","locator":"Tables 12 and 19","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"microsoft","sourceTitle":"MAI-Thinking-1 technical report"},"score":49,"scoreUnit":"percent","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf"},{"benchmarkId":"longbenchv2-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"longbenchv2-source-release-snapshot-version-not-specified:microsoft:TWljcm9zb2Z0IG93biBldmFsdWF0aW9uIHN1aXRlOyBTb25uZXQgbWF4IHJlYXNvbmluZzsgVGFibGVzIDEyLzE5LiBMb25nQmVuY2hWMiBjYXBwZWQgMjU2ayw0MDggcXVlc3Rpb25zOyBDb3JwdXNRQSBHUFQtNS40IGhpZ2gganVkZ2U7IDQgcnVucy4","id":"launch-microsoft-1425","ingestRunId":"launch-microsoft","modelId":"mai-thinking-1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs.","family":"LongBenchV2","locator":"Tables 12 and 19","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"microsoft","sourceTitle":"MAI-Thinking-1 technical report"},"score":61,"scoreUnit":"percent","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf"},{"benchmarkId":"longbenchv2-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"longbenchv2-source-release-snapshot-version-not-specified:microsoft:TWljcm9zb2Z0IG93biBldmFsdWF0aW9uIHN1aXRlOyBTb25uZXQgbWF4IHJlYXNvbmluZzsgVGFibGVzIDEyLzE5LiBMb25nQmVuY2hWMiBjYXBwZWQgMjU2ayw0MDggcXVlc3Rpb25zOyBDb3JwdXNRQSBHUFQtNS40IGhpZ2gganVkZ2U7IDQgcnVucy4","id":"launch-microsoft-1426","ingestRunId":"launch-microsoft","modelId":"claude-sonnet-4-6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs.","family":"LongBenchV2","locator":"Tables 12 and 19","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"microsoft","sourceTitle":"MAI-Thinking-1 technical report"},"score":66,"scoreUnit":"percent","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf"},{"benchmarkId":"corpusqa-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"corpusqa-source-release-snapshot-version-not-specified:microsoft:TWljcm9zb2Z0IG93biBldmFsdWF0aW9uIHN1aXRlOyBTb25uZXQgbWF4IHJlYXNvbmluZzsgVGFibGVzIDEyLzE5LiBMb25nQmVuY2hWMiBjYXBwZWQgMjU2ayw0MDggcXVlc3Rpb25zOyBDb3JwdXNRQSBHUFQtNS40IGhpZ2gganVkZ2U7IDQgcnVucy4","id":"launch-microsoft-1427","ingestRunId":"launch-microsoft","modelId":"mai-thinking-1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs.","family":"CorpusQA","locator":"Tables 12 and 19","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"microsoft","sourceTitle":"MAI-Thinking-1 technical report"},"score":82,"scoreUnit":"percent","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf"},{"benchmarkId":"corpusqa-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"corpusqa-source-release-snapshot-version-not-specified:microsoft:TWljcm9zb2Z0IG93biBldmFsdWF0aW9uIHN1aXRlOyBTb25uZXQgbWF4IHJlYXNvbmluZzsgVGFibGVzIDEyLzE5LiBMb25nQmVuY2hWMiBjYXBwZWQgMjU2ayw0MDggcXVlc3Rpb25zOyBDb3JwdXNRQSBHUFQtNS40IGhpZ2gganVkZ2U7IDQgcnVucy4","id":"launch-microsoft-1428","ingestRunId":"launch-microsoft","modelId":"claude-sonnet-4-6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs.","family":"CorpusQA","locator":"Tables 12 and 19","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"microsoft","sourceTitle":"MAI-Thinking-1 technical report"},"score":79,"scoreUnit":"percent","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf"},{"benchmarkId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1692","ingestRunId":"launch-minimax","modelId":"minimax-m3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"SWE-bench Verified","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":80.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1693","ingestRunId":"launch-minimax","modelId":"minimax-m2.7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"SWE-bench Verified","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":79.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1694","ingestRunId":"launch-minimax","modelId":"claude-opus-4-7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"SWE-bench Verified","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":87.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1695","ingestRunId":"launch-minimax","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"SWE-bench Verified","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":82.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1696","ingestRunId":"launch-minimax","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"SWE-bench Verified","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":80.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1697","ingestRunId":"launch-minimax","modelId":"claude-sonnet-4-6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"SWE-bench Verified","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":79.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1698","ingestRunId":"launch-minimax","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"SWE-bench Verified","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":80.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"swe-bench-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"swe-bench-pro-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1699","ingestRunId":"launch-minimax","modelId":"minimax-m3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"SWE-bench Pro","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":59,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"swe-bench-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"swe-bench-pro-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1700","ingestRunId":"launch-minimax","modelId":"minimax-m2.7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"SWE-bench Pro","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":56.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"swe-bench-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"swe-bench-pro-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1701","ingestRunId":"launch-minimax","modelId":"claude-opus-4-7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"SWE-bench Pro","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":64.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"swe-bench-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"swe-bench-pro-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1702","ingestRunId":"launch-minimax","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"SWE-bench Pro","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":58.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"swe-bench-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"swe-bench-pro-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1703","ingestRunId":"launch-minimax","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"SWE-bench Pro","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":54.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"swe-bench-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"swe-bench-pro-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1704","ingestRunId":"launch-minimax","modelId":"glm-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"SWE-bench Pro","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":58.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"swe-bench-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"swe-bench-pro-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1705","ingestRunId":"launch-minimax","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"SWE-bench Pro","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":58.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"terminal-bench-2-1-pct","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1-pct:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1706","ingestRunId":"launch-minimax","modelId":"minimax-m3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"Terminal-Bench","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":66,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"terminal-bench-2-1-pct","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1-pct:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1707","ingestRunId":"launch-minimax","modelId":"minimax-m2.7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"Terminal-Bench","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":51.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"terminal-bench-2-1-pct","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1-pct:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1708","ingestRunId":"launch-minimax","modelId":"claude-opus-4-7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"Terminal-Bench","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":66.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"terminal-bench-2-1-pct","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1-pct:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1709","ingestRunId":"launch-minimax","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"Terminal-Bench","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":78.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"terminal-bench-2-1-pct","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1-pct:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1710","ingestRunId":"launch-minimax","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"Terminal-Bench","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":70.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"terminal-bench-2-1-pct","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1-pct:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1711","ingestRunId":"launch-minimax","modelId":"glm-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"Terminal-Bench","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":48.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"terminal-bench-2-1-pct","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1-pct:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1712","ingestRunId":"launch-minimax","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"Terminal-Bench","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":53.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"sweatlas-qna-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"sweatlas-qna-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1713","ingestRunId":"launch-minimax","modelId":"minimax-m3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"SWEAtlas-QnA","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":37.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"sweatlas-qna-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"sweatlas-qna-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1714","ingestRunId":"launch-minimax","modelId":"minimax-m2.7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"SWEAtlas-QnA","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":11.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"sweatlas-qna-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"sweatlas-qna-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1715","ingestRunId":"launch-minimax","modelId":"claude-opus-4-7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"SWEAtlas-QnA","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":45.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"sweatlas-qna-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"sweatlas-qna-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1716","ingestRunId":"launch-minimax","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"SWEAtlas-QnA","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":45.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"sweatlas-qna-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"sweatlas-qna-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1717","ingestRunId":"launch-minimax","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"SWEAtlas-QnA","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":13.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"sweatlas-qna-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"sweatlas-qna-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1718","ingestRunId":"launch-minimax","modelId":"claude-sonnet-4-6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"SWEAtlas-QnA","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":31.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"nl2repo-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"nl2repo-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1719","ingestRunId":"launch-minimax","modelId":"minimax-m3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"NL2Repo","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":42.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"nl2repo-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"nl2repo-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1720","ingestRunId":"launch-minimax","modelId":"minimax-m2.7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"NL2Repo","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":35,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"nl2repo-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"nl2repo-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1721","ingestRunId":"launch-minimax","modelId":"claude-opus-4-7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"NL2Repo","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":56.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"nl2repo-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"nl2repo-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1722","ingestRunId":"launch-minimax","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"NL2Repo","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":52.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"nl2repo-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"nl2repo-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1723","ingestRunId":"launch-minimax","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"NL2Repo","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":21.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"nl2repo-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"nl2repo-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1724","ingestRunId":"launch-minimax","modelId":"glm-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"NL2Repo","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":41,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"nl2repo-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"nl2repo-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1725","ingestRunId":"launch-minimax","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"NL2Repo","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":42.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"sweatlas-testwriting-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"sweatlas-testwriting-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1726","ingestRunId":"launch-minimax","modelId":"minimax-m3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"SWEAtlas-TestWriting","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":30.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"sweatlas-testwriting-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"sweatlas-testwriting-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1727","ingestRunId":"launch-minimax","modelId":"minimax-m2.7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"SWEAtlas-TestWriting","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":18.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"sweatlas-testwriting-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"sweatlas-testwriting-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1728","ingestRunId":"launch-minimax","modelId":"claude-opus-4-7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"SWEAtlas-TestWriting","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":38.21,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"sweatlas-testwriting-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"sweatlas-testwriting-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1729","ingestRunId":"launch-minimax","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"SWEAtlas-TestWriting","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":42.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"sweatlas-testwriting-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"sweatlas-testwriting-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1730","ingestRunId":"launch-minimax","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"SWEAtlas-TestWriting","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":29.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"sweatlas-testwriting-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"sweatlas-testwriting-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1731","ingestRunId":"launch-minimax","modelId":"claude-sonnet-4-6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"SWEAtlas-TestWriting","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":31.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"swe-fficiency-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"swe-fficiency-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1732","ingestRunId":"launch-minimax","modelId":"minimax-m3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"SWE-fficiency","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":34.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"swe-fficiency-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"swe-fficiency-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1733","ingestRunId":"launch-minimax","modelId":"minimax-m2.7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"SWE-fficiency","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":14,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"swe-fficiency-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"swe-fficiency-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1734","ingestRunId":"launch-minimax","modelId":"claude-opus-4-7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"SWE-fficiency","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":42.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"swe-fficiency-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"swe-fficiency-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1735","ingestRunId":"launch-minimax","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"SWE-fficiency","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":46.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"swe-fficiency-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"swe-fficiency-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1736","ingestRunId":"launch-minimax","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"SWE-fficiency","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":19.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"livesqlbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"livesqlbench-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1737","ingestRunId":"launch-minimax","modelId":"minimax-m3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"LiveSQLBench","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":40.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"livesqlbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"livesqlbench-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1738","ingestRunId":"launch-minimax","modelId":"minimax-m2.7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"LiveSQLBench","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":33.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"livesqlbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"livesqlbench-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1739","ingestRunId":"launch-minimax","modelId":"claude-opus-4-7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"LiveSQLBench","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":41,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"livesqlbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"livesqlbench-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1740","ingestRunId":"launch-minimax","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"LiveSQLBench","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":40.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"livesqlbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"livesqlbench-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1741","ingestRunId":"launch-minimax","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"LiveSQLBench","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":39.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"cl-bench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"cl-bench-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1742","ingestRunId":"launch-minimax","modelId":"minimax-m3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"CL-bench","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":20.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"cl-bench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"cl-bench-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1743","ingestRunId":"launch-minimax","modelId":"minimax-m2.7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"CL-bench","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":15.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"cl-bench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"cl-bench-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1744","ingestRunId":"launch-minimax","modelId":"claude-opus-4-7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"CL-bench","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":22.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"cl-bench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"cl-bench-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1745","ingestRunId":"launch-minimax","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"CL-bench","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":25.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"cl-bench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"cl-bench-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1746","ingestRunId":"launch-minimax","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"CL-bench","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":21.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"vibe-v2-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"vibe-v2-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1747","ingestRunId":"launch-minimax","modelId":"minimax-m3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"VIBE-V2","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":50.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"vibe-v2-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"vibe-v2-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1748","ingestRunId":"launch-minimax","modelId":"minimax-m2.7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"VIBE-V2","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":37.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"vibe-v2-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"vibe-v2-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1749","ingestRunId":"launch-minimax","modelId":"claude-opus-4-7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"VIBE-V2","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":55.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"vibe-v2-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"vibe-v2-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1750","ingestRunId":"launch-minimax","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"VIBE-V2","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":50.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"vibe-v2-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"vibe-v2-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1751","ingestRunId":"launch-minimax","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"VIBE-V2","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":28,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"vibe-v2-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"vibe-v2-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1752","ingestRunId":"launch-minimax","modelId":"claude-sonnet-4-6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"VIBE-V2","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":42.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"vibe-v2-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"vibe-v2-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1753","ingestRunId":"launch-minimax","modelId":"glm-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"VIBE-V2","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":48.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"vibe-v2-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"vibe-v2-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1754","ingestRunId":"launch-minimax","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"VIBE-V2","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":46,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"svg-bench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"svg-bench-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1755","ingestRunId":"launch-minimax","modelId":"minimax-m3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"SVG-Bench","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":63.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"svg-bench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"svg-bench-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1756","ingestRunId":"launch-minimax","modelId":"minimax-m2.7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"SVG-Bench","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":48,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"svg-bench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"svg-bench-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1757","ingestRunId":"launch-minimax","modelId":"claude-opus-4-7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"SVG-Bench","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":62.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"svg-bench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"svg-bench-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1758","ingestRunId":"launch-minimax","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"SVG-Bench","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":58.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"svg-bench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"svg-bench-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1759","ingestRunId":"launch-minimax","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"SVG-Bench","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":59.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"svg-bench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"svg-bench-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1760","ingestRunId":"launch-minimax","modelId":"claude-sonnet-4-6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"SVG-Bench","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":64.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"svg-bench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"svg-bench-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1761","ingestRunId":"launch-minimax","modelId":"glm-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"SVG-Bench","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":56.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"svg-bench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"svg-bench-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1762","ingestRunId":"launch-minimax","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"SVG-Bench","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":60,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"posttrainbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"posttrainbench-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1763","ingestRunId":"launch-minimax","modelId":"minimax-m3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"PostTrainBench","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":37.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"posttrainbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"posttrainbench-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1764","ingestRunId":"launch-minimax","modelId":"minimax-m2.7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"PostTrainBench","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":13.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"posttrainbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"posttrainbench-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1765","ingestRunId":"launch-minimax","modelId":"claude-opus-4-7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"PostTrainBench","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":42.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"posttrainbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"posttrainbench-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1766","ingestRunId":"launch-minimax","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"PostTrainBench","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":39.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"posttrainbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"posttrainbench-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1767","ingestRunId":"launch-minimax","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"PostTrainBench","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":15.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"kernelbench-hard-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"kernelbench-hard-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1768","ingestRunId":"launch-minimax","modelId":"minimax-m3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"KernelBench Hard","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":28.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"kernelbench-hard-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"kernelbench-hard-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1769","ingestRunId":"launch-minimax","modelId":"minimax-m2.7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"KernelBench Hard","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":10.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"kernelbench-hard-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"kernelbench-hard-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1770","ingestRunId":"launch-minimax","modelId":"claude-opus-4-7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"KernelBench Hard","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":30.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"kernelbench-hard-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"kernelbench-hard-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1771","ingestRunId":"launch-minimax","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"KernelBench Hard","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":20.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"kernelbench-hard-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"kernelbench-hard-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1772","ingestRunId":"launch-minimax","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"KernelBench Hard","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":18.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"paperbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"paperbench-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1773","ingestRunId":"launch-minimax","modelId":"minimax-m3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"PaperBench","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":52.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"paperbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"paperbench-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1774","ingestRunId":"launch-minimax","modelId":"minimax-m2.7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"PaperBench","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":30.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"paperbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"paperbench-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1775","ingestRunId":"launch-minimax","modelId":"claude-opus-4-7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"PaperBench","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":58.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"paperbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"paperbench-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1776","ingestRunId":"launch-minimax","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"PaperBench","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":57.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"paperbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"paperbench-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1777","ingestRunId":"launch-minimax","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"PaperBench","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":46.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"browsecomp-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"browsecomp-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1778","ingestRunId":"launch-minimax","modelId":"minimax-m3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"BrowseComp","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":83.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"browsecomp-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"browsecomp-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1779","ingestRunId":"launch-minimax","modelId":"minimax-m2.7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"BrowseComp","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":76.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"browsecomp-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"browsecomp-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1780","ingestRunId":"launch-minimax","modelId":"claude-opus-4-7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"BrowseComp","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":79.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"browsecomp-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"browsecomp-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1781","ingestRunId":"launch-minimax","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"BrowseComp","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":84.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"browsecomp-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"browsecomp-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1782","ingestRunId":"launch-minimax","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"BrowseComp","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":85.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"browsecomp-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"browsecomp-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1783","ingestRunId":"launch-minimax","modelId":"claude-sonnet-4-6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"BrowseComp","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":74.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"browsecomp-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"browsecomp-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1784","ingestRunId":"launch-minimax","modelId":"glm-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"BrowseComp","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":79.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"browsecomp-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"browsecomp-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1785","ingestRunId":"launch-minimax","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"BrowseComp","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":83.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"draco-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"draco-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1786","ingestRunId":"launch-minimax","modelId":"minimax-m3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"DRACO","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":73.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"draco-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"draco-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1787","ingestRunId":"launch-minimax","modelId":"minimax-m2.7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"DRACO","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":66.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"draco-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"draco-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1788","ingestRunId":"launch-minimax","modelId":"claude-opus-4-7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"DRACO","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":77.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"draco-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"draco-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1789","ingestRunId":"launch-minimax","modelId":"claude-sonnet-4-6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"DRACO","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":75.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"gdpval-rubrics-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gdpval-rubrics-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1790","ingestRunId":"launch-minimax","modelId":"minimax-m3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"GDPval rubrics","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":74.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"gdpval-rubrics-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gdpval-rubrics-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1791","ingestRunId":"launch-minimax","modelId":"minimax-m2.7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"GDPval rubrics","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":66.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"gdpval-rubrics-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gdpval-rubrics-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1792","ingestRunId":"launch-minimax","modelId":"claude-opus-4-7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"GDPval rubrics","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":79.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"gdpval-rubrics-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gdpval-rubrics-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1793","ingestRunId":"launch-minimax","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"GDPval rubrics","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":80.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"gdpval-rubrics-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gdpval-rubrics-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1794","ingestRunId":"launch-minimax","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"GDPval rubrics","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":57.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"gdpval-rubrics-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gdpval-rubrics-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1795","ingestRunId":"launch-minimax","modelId":"claude-sonnet-4-6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"GDPval rubrics","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":75.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"gdpval-rubrics-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gdpval-rubrics-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1796","ingestRunId":"launch-minimax","modelId":"glm-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"GDPval rubrics","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":68.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"gdpval-rubrics-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gdpval-rubrics-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1797","ingestRunId":"launch-minimax","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"GDPval rubrics","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":65.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"bankertoolbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"bankertoolbench-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1798","ingestRunId":"launch-minimax","modelId":"minimax-m3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"BankerToolBench","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":76.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"bankertoolbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"bankertoolbench-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1799","ingestRunId":"launch-minimax","modelId":"minimax-m2.7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"BankerToolBench","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":63.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"bankertoolbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"bankertoolbench-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1800","ingestRunId":"launch-minimax","modelId":"claude-opus-4-7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"BankerToolBench","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":81.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"bankertoolbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"bankertoolbench-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1801","ingestRunId":"launch-minimax","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"BankerToolBench","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":70,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"bankertoolbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"bankertoolbench-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1802","ingestRunId":"launch-minimax","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"BankerToolBench","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":67,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"officeqa-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"officeqa-pro-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1803","ingestRunId":"launch-minimax","modelId":"minimax-m3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"OfficeQA Pro","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":45.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"officeqa-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"officeqa-pro-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1804","ingestRunId":"launch-minimax","modelId":"claude-opus-4-7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"OfficeQA Pro","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":43.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"officeqa-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"officeqa-pro-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1805","ingestRunId":"launch-minimax","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"OfficeQA Pro","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":52.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"officeqa-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"officeqa-pro-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1806","ingestRunId":"launch-minimax","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"OfficeQA Pro","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":18.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"spreadsheetbench-v1-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"spreadsheetbench-v1-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1807","ingestRunId":"launch-minimax","modelId":"minimax-m3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"SpreadsheetBench v1","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":89.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"spreadsheetbench-v1-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"spreadsheetbench-v1-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1808","ingestRunId":"launch-minimax","modelId":"minimax-m2.7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"SpreadsheetBench v1","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":84.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"spreadsheetbench-v1-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"spreadsheetbench-v1-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1809","ingestRunId":"launch-minimax","modelId":"claude-opus-4-7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"SpreadsheetBench v1","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":88.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"spreadsheetbench-v1-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"spreadsheetbench-v1-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1810","ingestRunId":"launch-minimax","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"SpreadsheetBench v1","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":88.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"spreadsheetbench-v1-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"spreadsheetbench-v1-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1811","ingestRunId":"launch-minimax","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"SpreadsheetBench v1","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":56.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"spreadsheetbench-v1-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"spreadsheetbench-v1-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1812","ingestRunId":"launch-minimax","modelId":"glm-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"SpreadsheetBench v1","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":85.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"spreadsheetbench-v1-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"spreadsheetbench-v1-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1813","ingestRunId":"launch-minimax","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"SpreadsheetBench v1","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":84.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"loca-bench-256k-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"loca-bench-256k-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1814","ingestRunId":"launch-minimax","modelId":"minimax-m3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"LOCA-Bench 256k","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":49.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"loca-bench-256k-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"loca-bench-256k-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1815","ingestRunId":"launch-minimax","modelId":"claude-opus-4-7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"LOCA-Bench 256k","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":57,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"mcpatlas-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mcpatlas-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1816","ingestRunId":"launch-minimax","modelId":"minimax-m3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"MCPAtlas","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":74.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"mcpatlas-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mcpatlas-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1817","ingestRunId":"launch-minimax","modelId":"minimax-m2.7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"MCPAtlas","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":49.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"mcpatlas-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mcpatlas-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1818","ingestRunId":"launch-minimax","modelId":"claude-opus-4-7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"MCPAtlas","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":77,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"mcpatlas-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mcpatlas-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1819","ingestRunId":"launch-minimax","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"MCPAtlas","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":75.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"mcpatlas-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mcpatlas-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1820","ingestRunId":"launch-minimax","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"MCPAtlas","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":69.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"mcpatlas-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mcpatlas-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1821","ingestRunId":"launch-minimax","modelId":"claude-sonnet-4-6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"MCPAtlas","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":61.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"mcpatlas-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mcpatlas-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1822","ingestRunId":"launch-minimax","modelId":"glm-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"MCPAtlas","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":71.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"mcpatlas-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mcpatlas-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1823","ingestRunId":"launch-minimax","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"MCPAtlas","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":66.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"apex-agents-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"apex-agents-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1824","ingestRunId":"launch-minimax","modelId":"minimax-m3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"APEX-Agents","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":27.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"apex-agents-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"apex-agents-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1825","ingestRunId":"launch-minimax","modelId":"minimax-m2.7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"APEX-Agents","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":5.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"apex-agents-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"apex-agents-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1826","ingestRunId":"launch-minimax","modelId":"claude-opus-4-7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"APEX-Agents","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":37.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"apex-agents-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"apex-agents-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1827","ingestRunId":"launch-minimax","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"APEX-Agents","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":41.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"apex-agents-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"apex-agents-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1828","ingestRunId":"launch-minimax","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"APEX-Agents","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":33.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"apex-agents-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"apex-agents-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1829","ingestRunId":"launch-minimax","modelId":"claude-sonnet-4-6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"APEX-Agents","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":26.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"claw-eval-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"claw-eval-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1830","ingestRunId":"launch-minimax","modelId":"minimax-m3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"Claw-Eval","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":74.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"claw-eval-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"claw-eval-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1831","ingestRunId":"launch-minimax","modelId":"minimax-m2.7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"Claw-Eval","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":49.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"claw-eval-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"claw-eval-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1832","ingestRunId":"launch-minimax","modelId":"claude-opus-4-7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"Claw-Eval","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":71.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"claw-eval-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"claw-eval-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1833","ingestRunId":"launch-minimax","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"Claw-Eval","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":57.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"claw-eval-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"claw-eval-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1834","ingestRunId":"launch-minimax","modelId":"claude-sonnet-4-6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"Claw-Eval","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":68.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"claw-eval-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"claw-eval-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1835","ingestRunId":"launch-minimax","modelId":"glm-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"Claw-Eval","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":62.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"claw-eval-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"claw-eval-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1836","ingestRunId":"launch-minimax","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"Claw-Eval","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":61.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"osworld-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"osworld-verified-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1837","ingestRunId":"launch-minimax","modelId":"minimax-m3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"OSWorld","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":75.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"osworld-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"osworld-verified-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1838","ingestRunId":"launch-minimax","modelId":"claude-opus-4-7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"OSWorld","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":82.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"osworld-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"osworld-verified-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1839","ingestRunId":"launch-minimax","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"OSWorld","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":78.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"osworld-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"osworld-verified-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1840","ingestRunId":"launch-minimax","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"OSWorld","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":76.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"osworld-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"osworld-verified-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1841","ingestRunId":"launch-minimax","modelId":"claude-sonnet-4-6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"OSWorld","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":72.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"osworld-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"osworld-verified-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1842","ingestRunId":"launch-minimax","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"OSWorld","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":73.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"omnidocbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"omnidocbench-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1843","ingestRunId":"launch-minimax","modelId":"minimax-m3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"OmniDocBench","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":91.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"omnidocbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"omnidocbench-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1844","ingestRunId":"launch-minimax","modelId":"claude-opus-4-7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"OmniDocBench","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":89.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"omnidocbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"omnidocbench-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1845","ingestRunId":"launch-minimax","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"OmniDocBench","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":87.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"omnidocbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"omnidocbench-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1846","ingestRunId":"launch-minimax","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"OmniDocBench","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":88.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"omnidocbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"omnidocbench-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1847","ingestRunId":"launch-minimax","modelId":"claude-sonnet-4-6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"OmniDocBench","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":86.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"mmmu-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mmmu-pro-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1848","ingestRunId":"launch-minimax","modelId":"minimax-m3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"MMMU-Pro","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":78.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"mmmu-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mmmu-pro-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1849","ingestRunId":"launch-minimax","modelId":"claude-opus-4-7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"MMMU-Pro","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":77,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"mmmu-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mmmu-pro-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1850","ingestRunId":"launch-minimax","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"MMMU-Pro","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":81.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"mmmu-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mmmu-pro-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1851","ingestRunId":"launch-minimax","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"MMMU-Pro","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":80.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"mmmu-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mmmu-pro-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1852","ingestRunId":"launch-minimax","modelId":"claude-sonnet-4-6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"MMMU-Pro","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":74.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"mmmu-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mmmu-pro-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1853","ingestRunId":"launch-minimax","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"MMMU-Pro","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":79.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"videommmu-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"videommmu-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1854","ingestRunId":"launch-minimax","modelId":"minimax-m3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"VideoMMMU","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":84.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"videommmu-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"videommmu-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1855","ingestRunId":"launch-minimax","modelId":"claude-opus-4-7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"VideoMMMU","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":83,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"videommmu-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"videommmu-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1856","ingestRunId":"launch-minimax","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"VideoMMMU","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":86.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"videommmu-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"videommmu-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1857","ingestRunId":"launch-minimax","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"VideoMMMU","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":87.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"videomme-with-subtitles-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"videomme-with-subtitles-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1858","ingestRunId":"launch-minimax","modelId":"minimax-m3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"VideoMME with subtitles","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":85.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"videomme-with-subtitles-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"videomme-with-subtitles-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1859","ingestRunId":"launch-minimax","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"VideoMME with subtitles","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":89.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"videomme-with-subtitles-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"videomme-with-subtitles-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBsYXVuY2ggZmlndXJlIG1ldGhvZG9sb2d5LiBTV0UgaW50ZXJuYWwgQ2xhdWRlQ29kZSA0IHJ1bnM7IFRlcm1pbmFsIDhDUFUxNkdCMmgxMjhrIFRlcm1pbnVzMjsgUGFwZXIvUG9zdFRyYWluIFJhbHBoLWxvb3AxMmg7IEdEUHZhbCBpbnRlcm5hbCBydWJyaWMgZ3JhZGVyOyBCcm93c2VDb21wIGRpc2NhcmQ2NGs7IE9TV29ybGQzNjF0YXNrcy4gQ29tcGFyYXRvciBwcm92aWRlci9sZWFkZXJib2FyZCBzY29yZXMgd2hlcmUgZm9vdG5vdGVkLg","id":"launch-minimax-1860","ingestRunId":"launch-minimax","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted.","family":"VideoMME with subtitles","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":87.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"yc-bench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"yc-bench-source-release-snapshot-version-not-specified:minimax:TGF1bmNoIGZpZ3VyZTsgYWdlbnQgZmluYWwgYXNzZXRz","id":"launch-minimax-1861","ingestRunId":"launch-minimax","modelId":"minimax-m3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch figure; agent final assets","family":"YC-Bench","locator":"figures/benchmark.jpeg","metric":"USD","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":2100000,"scoreUnit":"index","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"yc-bench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"yc-bench-source-release-snapshot-version-not-specified:minimax:TGF1bmNoIGZpZ3VyZTsgYWdlbnQgZmluYWwgYXNzZXRz","id":"launch-minimax-1862","ingestRunId":"launch-minimax","modelId":"minimax-m2.7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch figure; agent final assets","family":"YC-Bench","locator":"figures/benchmark.jpeg","metric":"USD","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":0,"scoreUnit":"index","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"yc-bench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"yc-bench-source-release-snapshot-version-not-specified:minimax:TGF1bmNoIGZpZ3VyZTsgYWdlbnQgZmluYWwgYXNzZXRz","id":"launch-minimax-1863","ingestRunId":"launch-minimax","modelId":"claude-opus-4-7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch figure; agent final assets","family":"YC-Bench","locator":"figures/benchmark.jpeg","metric":"USD","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":2200000,"scoreUnit":"index","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"yc-bench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"yc-bench-source-release-snapshot-version-not-specified:minimax:TGF1bmNoIGZpZ3VyZTsgYWdlbnQgZmluYWwgYXNzZXRz","id":"launch-minimax-1864","ingestRunId":"launch-minimax","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch figure; agent final assets","family":"YC-Bench","locator":"figures/benchmark.jpeg","metric":"USD","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":1300000,"scoreUnit":"index","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"yc-bench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"yc-bench-source-release-snapshot-version-not-specified:minimax:TGF1bmNoIGZpZ3VyZTsgYWdlbnQgZmluYWwgYXNzZXRz","id":"launch-minimax-1865","ingestRunId":"launch-minimax","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch figure; agent final assets","family":"YC-Bench","locator":"figures/benchmark.jpeg","metric":"USD","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":1100000,"scoreUnit":"index","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"yc-bench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"yc-bench-source-release-snapshot-version-not-specified:minimax:TGF1bmNoIGZpZ3VyZTsgYWdlbnQgZmluYWwgYXNzZXRz","id":"launch-minimax-1866","ingestRunId":"launch-minimax","modelId":"claude-sonnet-4-6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch figure; agent final assets","family":"YC-Bench","locator":"figures/benchmark.jpeg","metric":"USD","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":100000,"scoreUnit":"index","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"imo-2025-source-release-snapshot-version-not-specified-points-42","evidenceKind":"lab_self_report","harnessId":"imo-2025-source-release-snapshot-version-not-specified-points-42:minimax:TWluaU1heCBwb2ludHMgb3V0IG9mNDI7IGNvbXBhcmF0b3IgcGVyY2VudGFnZXMgYXMgcHJpbnRlZA","id":"launch-minimax-1867","ingestRunId":"launch-minimax","modelId":"minimax-m3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"MiniMax points out of42; comparator percentages as printed","family":"IMO 2025","locator":"figures/benchmark.jpeg","metric":"points / 42","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":35,"scoreUnit":"index","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"imo-2025-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"imo-2025-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBwb2ludHMgb3V0IG9mNDI7IGNvbXBhcmF0b3IgcGVyY2VudGFnZXMgYXMgcHJpbnRlZA","id":"launch-minimax-1868","ingestRunId":"launch-minimax","modelId":"claude-opus-4-7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"MiniMax points out of42; comparator percentages as printed","family":"IMO 2025","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":17.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"imo-2025-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"imo-2025-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBwb2ludHMgb3V0IG9mNDI7IGNvbXBhcmF0b3IgcGVyY2VudGFnZXMgYXMgcHJpbnRlZA","id":"launch-minimax-1869","ingestRunId":"launch-minimax","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"MiniMax points out of42; comparator percentages as printed","family":"IMO 2025","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":76.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"imo-2025-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"imo-2025-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBwb2ludHMgb3V0IG9mNDI7IGNvbXBhcmF0b3IgcGVyY2VudGFnZXMgYXMgcHJpbnRlZA","id":"launch-minimax-1870","ingestRunId":"launch-minimax","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"MiniMax points out of42; comparator percentages as printed","family":"IMO 2025","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":42.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"usamo-2026-source-release-snapshot-version-not-specified-points-42","evidenceKind":"lab_self_report","harnessId":"usamo-2026-source-release-snapshot-version-not-specified-points-42:minimax:TWluaU1heCBwb2ludHMgb3V0IG9mNDI7IGNvbXBhcmF0b3IgcGVyY2VudGFnZXMgYXMgcHJpbnRlZA","id":"launch-minimax-1871","ingestRunId":"launch-minimax","modelId":"minimax-m3","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"MiniMax points out of42; comparator percentages as printed","family":"USAMO 2026","locator":"figures/benchmark.jpeg","metric":"points / 42","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":36,"scoreUnit":"index","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"usamo-2026-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"usamo-2026-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBwb2ludHMgb3V0IG9mNDI7IGNvbXBhcmF0b3IgcGVyY2VudGFnZXMgYXMgcHJpbnRlZA","id":"launch-minimax-1872","ingestRunId":"launch-minimax","modelId":"claude-opus-4-7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"MiniMax points out of42; comparator percentages as printed","family":"USAMO 2026","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":52.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"usamo-2026-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"usamo-2026-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBwb2ludHMgb3V0IG9mNDI7IGNvbXBhcmF0b3IgcGVyY2VudGFnZXMgYXMgcHJpbnRlZA","id":"launch-minimax-1873","ingestRunId":"launch-minimax","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"MiniMax points out of42; comparator percentages as printed","family":"USAMO 2026","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":98.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"usamo-2026-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"usamo-2026-source-release-snapshot-version-not-specified:minimax:TWluaU1heCBwb2ludHMgb3V0IG9mNDI7IGNvbXBhcmF0b3IgcGVyY2VudGFnZXMgYXMgcHJpbnRlZA","id":"launch-minimax-1874","ingestRunId":"launch-minimax","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"MiniMax points out of42; comparator percentages as printed","family":"USAMO 2026","locator":"figures/benchmark.jpeg","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"minimax","sourceTitle":"MiniMax M3 model card benchmark figure"},"score":74.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3"},{"benchmarkId":"aime-2025-avg-16-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"aime-2025-avg-16-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IFNvbm5ldDQuNiBleHRlcm5hbCBBUEkgdHJ1bmNhdGlvbiBjYXZlYXQu","id":"launch-mistral-1442","ingestRunId":"launch-mistral","modelId":"mistral-medium-3.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Maximum reasoning; Sonnet4.6 external API truncation caveat.","family":"AIME","locator":"images/image1.png","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":86.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"aime-2025-avg-16-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"aime-2025-avg-16-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IFNvbm5ldDQuNiBleHRlcm5hbCBBUEkgdHJ1bmNhdGlvbiBjYXZlYXQu","id":"launch-mistral-1443","ingestRunId":"launch-mistral","modelId":"claude-sonnet-4-5-20250929","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Maximum reasoning; Sonnet4.6 external API truncation caveat.","family":"AIME","locator":"images/image1.png","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":86.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"aime-2025-avg-16-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"aime-2025-avg-16-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IFNvbm5ldDQuNiBleHRlcm5hbCBBUEkgdHJ1bmNhdGlvbiBjYXZlYXQu","id":"launch-mistral-1444","ingestRunId":"launch-mistral","modelId":"claude-sonnet-4-6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Maximum reasoning; Sonnet4.6 external API truncation caveat.","family":"AIME","locator":"images/image1.png","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":86.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"aime-2025-avg-16-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"aime-2025-avg-16-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IFNvbm5ldDQuNiBleHRlcm5hbCBBUEkgdHJ1bmNhdGlvbiBjYXZlYXQu","id":"launch-mistral-1445","ingestRunId":"launch-mistral","modelId":"qwen3.5-397b-a17b","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Maximum reasoning; Sonnet4.6 external API truncation caveat.","family":"AIME","locator":"images/image1.png","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":83.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"aime-2025-avg-16-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"aime-2025-avg-16-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IFNvbm5ldDQuNiBleHRlcm5hbCBBUEkgdHJ1bmNhdGlvbiBjYXZlYXQu","id":"launch-mistral-1446","ingestRunId":"launch-mistral","modelId":"glm-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Maximum reasoning; Sonnet4.6 external API truncation caveat.","family":"AIME","locator":"images/image1.png","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":87.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"allenai-ifbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"allenai-ifbench-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IFNvbm5ldDQuNiBleHRlcm5hbCBBUEkgdHJ1bmNhdGlvbiBjYXZlYXQu","id":"launch-mistral-1447","ingestRunId":"launch-mistral","modelId":"mistral-medium-3.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"Maximum reasoning; Sonnet4.6 external API truncation caveat.","family":"AllenAI IFBench","locator":"images/image1.png","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":69,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"allenai-ifbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"allenai-ifbench-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IFNvbm5ldDQuNiBleHRlcm5hbCBBUEkgdHJ1bmNhdGlvbiBjYXZlYXQu","id":"launch-mistral-1448","ingestRunId":"launch-mistral","modelId":"claude-sonnet-4-5-20250929","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"Maximum reasoning; Sonnet4.6 external API truncation caveat.","family":"AllenAI IFBench","locator":"images/image1.png","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":55.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"allenai-ifbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"allenai-ifbench-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IFNvbm5ldDQuNiBleHRlcm5hbCBBUEkgdHJ1bmNhdGlvbiBjYXZlYXQu","id":"launch-mistral-1449","ingestRunId":"launch-mistral","modelId":"claude-sonnet-4-6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"Maximum reasoning; Sonnet4.6 external API truncation caveat.","family":"AllenAI IFBench","locator":"images/image1.png","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":57.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"allenai-ifbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"allenai-ifbench-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IFNvbm5ldDQuNiBleHRlcm5hbCBBUEkgdHJ1bmNhdGlvbiBjYXZlYXQu","id":"launch-mistral-1450","ingestRunId":"launch-mistral","modelId":"qwen3.5-397b-a17b","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"Maximum reasoning; Sonnet4.6 external API truncation caveat.","family":"AllenAI IFBench","locator":"images/image1.png","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":76.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"allenai-ifbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"allenai-ifbench-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IFNvbm5ldDQuNiBleHRlcm5hbCBBUEkgdHJ1bmNhdGlvbiBjYXZlYXQu","id":"launch-mistral-1451","ingestRunId":"launch-mistral","modelId":"glm-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"Maximum reasoning; Sonnet4.6 external API truncation caveat.","family":"AllenAI IFBench","locator":"images/image1.png","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":67,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"collie-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"collie-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IFNvbm5ldDQuNiBleHRlcm5hbCBBUEkgdHJ1bmNhdGlvbiBjYXZlYXQu","id":"launch-mistral-1452","ingestRunId":"launch-mistral","modelId":"mistral-medium-3.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"Maximum reasoning; Sonnet4.6 external API truncation caveat.","family":"Collie","locator":"images/image1.png","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":95.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"collie-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"collie-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IFNvbm5ldDQuNiBleHRlcm5hbCBBUEkgdHJ1bmNhdGlvbiBjYXZlYXQu","id":"launch-mistral-1453","ingestRunId":"launch-mistral","modelId":"claude-sonnet-4-5-20250929","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"Maximum reasoning; Sonnet4.6 external API truncation caveat.","family":"Collie","locator":"images/image1.png","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":90.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"collie-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"collie-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IFNvbm5ldDQuNiBleHRlcm5hbCBBUEkgdHJ1bmNhdGlvbiBjYXZlYXQu","id":"launch-mistral-1454","ingestRunId":"launch-mistral","modelId":"claude-sonnet-4-6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"Maximum reasoning; Sonnet4.6 external API truncation caveat.","family":"Collie","locator":"images/image1.png","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":67.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"collie-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"collie-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IFNvbm5ldDQuNiBleHRlcm5hbCBBUEkgdHJ1bmNhdGlvbiBjYXZlYXQu","id":"launch-mistral-1455","ingestRunId":"launch-mistral","modelId":"qwen3.5-397b-a17b","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"Maximum reasoning; Sonnet4.6 external API truncation caveat.","family":"Collie","locator":"images/image1.png","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":88.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"collie-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"collie-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IFNvbm5ldDQuNiBleHRlcm5hbCBBUEkgdHJ1bmNhdGlvbiBjYXZlYXQu","id":"launch-mistral-1456","ingestRunId":"launch-mistral","modelId":"glm-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"Maximum reasoning; Sonnet4.6 external API truncation caveat.","family":"Collie","locator":"images/image1.png","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":86.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"beyondaime-avg-16-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"beyondaime-avg-16-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IFNvbm5ldDQuNiBleHRlcm5hbCBBUEkgdHJ1bmNhdGlvbiBjYXZlYXQu","id":"launch-mistral-1457","ingestRunId":"launch-mistral","modelId":"mistral-medium-3.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Maximum reasoning; Sonnet4.6 external API truncation caveat.","family":"BeyondAIME avg@16","locator":"images/image1.png","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":66.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"beyondaime-avg-16-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"beyondaime-avg-16-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IFNvbm5ldDQuNiBleHRlcm5hbCBBUEkgdHJ1bmNhdGlvbiBjYXZlYXQu","id":"launch-mistral-1458","ingestRunId":"launch-mistral","modelId":"claude-sonnet-4-5-20250929","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Maximum reasoning; Sonnet4.6 external API truncation caveat.","family":"BeyondAIME avg@16","locator":"images/image1.png","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":59.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"beyondaime-avg-16-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"beyondaime-avg-16-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IFNvbm5ldDQuNiBleHRlcm5hbCBBUEkgdHJ1bmNhdGlvbiBjYXZlYXQu","id":"launch-mistral-1459","ingestRunId":"launch-mistral","modelId":"claude-sonnet-4-6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Maximum reasoning; Sonnet4.6 external API truncation caveat.","family":"BeyondAIME avg@16","locator":"images/image1.png","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":47.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"beyondaime-avg-16-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"beyondaime-avg-16-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IFNvbm5ldDQuNiBleHRlcm5hbCBBUEkgdHJ1bmNhdGlvbiBjYXZlYXQu","id":"launch-mistral-1460","ingestRunId":"launch-mistral","modelId":"qwen3.5-397b-a17b","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Maximum reasoning; Sonnet4.6 external API truncation caveat.","family":"BeyondAIME avg@16","locator":"images/image1.png","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":72.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct:mistral:TWF4aW11bSByZWFzb25pbmc7IHRhdTMgNCB0cmlhbHMsIEdQVDUuMiBsb3cgc2ltdWxhdG9yIGV4Y2VwdCBTaWVycmEgcmVwb3J0ZWQgY29tcGFyaXNvbnM7IEJyb3dzZUNvbXAgZGlzY2FyZC1hbGwgY29udGV4dCBhdDEwMGsu","id":"launch-mistral-1461","ingestRunId":"launch-mistral","modelId":"mistral-medium-3.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k.","family":"SWE-bench Verified","locator":"images/image4.png","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":77.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct:mistral:TWF4aW11bSByZWFzb25pbmc7IHRhdTMgNCB0cmlhbHMsIEdQVDUuMiBsb3cgc2ltdWxhdG9yIGV4Y2VwdCBTaWVycmEgcmVwb3J0ZWQgY29tcGFyaXNvbnM7IEJyb3dzZUNvbXAgZGlzY2FyZC1hbGwgY29udGV4dCBhdDEwMGsu","id":"launch-mistral-1462","ingestRunId":"launch-mistral","modelId":"claude-sonnet-4-5-20250929","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k.","family":"SWE-bench Verified","locator":"images/image4.png","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":77.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct:mistral:TWF4aW11bSByZWFzb25pbmc7IHRhdTMgNCB0cmlhbHMsIEdQVDUuMiBsb3cgc2ltdWxhdG9yIGV4Y2VwdCBTaWVycmEgcmVwb3J0ZWQgY29tcGFyaXNvbnM7IEJyb3dzZUNvbXAgZGlzY2FyZC1hbGwgY29udGV4dCBhdDEwMGsu","id":"launch-mistral-1463","ingestRunId":"launch-mistral","modelId":"claude-sonnet-4-6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k.","family":"SWE-bench Verified","locator":"images/image4.png","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":79.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct:mistral:TWF4aW11bSByZWFzb25pbmc7IHRhdTMgNCB0cmlhbHMsIEdQVDUuMiBsb3cgc2ltdWxhdG9yIGV4Y2VwdCBTaWVycmEgcmVwb3J0ZWQgY29tcGFyaXNvbnM7IEJyb3dzZUNvbXAgZGlzY2FyZC1hbGwgY29udGV4dCBhdDEwMGsu","id":"launch-mistral-1464","ingestRunId":"launch-mistral","modelId":"glm-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k.","family":"SWE-bench Verified","locator":"images/image4.png","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":80.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"swe-bench-verified-source-release-snapshot-version-not-specified-pct:mistral:TWF4aW11bSByZWFzb25pbmc7IHRhdTMgNCB0cmlhbHMsIEdQVDUuMiBsb3cgc2ltdWxhdG9yIGV4Y2VwdCBTaWVycmEgcmVwb3J0ZWQgY29tcGFyaXNvbnM7IEJyb3dzZUNvbXAgZGlzY2FyZC1hbGwgY29udGV4dCBhdDEwMGsu","id":"launch-mistral-1465","ingestRunId":"launch-mistral","modelId":"qwen3.5-397b-a17b","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k.","family":"SWE-bench Verified","locator":"images/image4.png","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":76.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"tau3-telecom-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"tau3-telecom-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IHRhdTMgNCB0cmlhbHMsIEdQVDUuMiBsb3cgc2ltdWxhdG9yIGV4Y2VwdCBTaWVycmEgcmVwb3J0ZWQgY29tcGFyaXNvbnM7IEJyb3dzZUNvbXAgZGlzY2FyZC1hbGwgY29udGV4dCBhdDEwMGsu","id":"launch-mistral-1466","ingestRunId":"launch-mistral","modelId":"mistral-medium-3.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k.","family":"tau3 Telecom","locator":"images/image4.png","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":91.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"tau3-telecom-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"tau3-telecom-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IHRhdTMgNCB0cmlhbHMsIEdQVDUuMiBsb3cgc2ltdWxhdG9yIGV4Y2VwdCBTaWVycmEgcmVwb3J0ZWQgY29tcGFyaXNvbnM7IEJyb3dzZUNvbXAgZGlzY2FyZC1hbGwgY29udGV4dCBhdDEwMGsu","id":"launch-mistral-1467","ingestRunId":"launch-mistral","modelId":"claude-sonnet-4-5-20250929","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k.","family":"tau3 Telecom","locator":"images/image4.png","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":84.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"tau3-telecom-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"tau3-telecom-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IHRhdTMgNCB0cmlhbHMsIEdQVDUuMiBsb3cgc2ltdWxhdG9yIGV4Y2VwdCBTaWVycmEgcmVwb3J0ZWQgY29tcGFyaXNvbnM7IEJyb3dzZUNvbXAgZGlzY2FyZC1hbGwgY29udGV4dCBhdDEwMGsu","id":"launch-mistral-1468","ingestRunId":"launch-mistral","modelId":"claude-sonnet-4-6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k.","family":"tau3 Telecom","locator":"images/image4.png","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":70.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"tau3-telecom-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"tau3-telecom-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IHRhdTMgNCB0cmlhbHMsIEdQVDUuMiBsb3cgc2ltdWxhdG9yIGV4Y2VwdCBTaWVycmEgcmVwb3J0ZWQgY29tcGFyaXNvbnM7IEJyb3dzZUNvbXAgZGlzY2FyZC1hbGwgY29udGV4dCBhdDEwMGsu","id":"launch-mistral-1469","ingestRunId":"launch-mistral","modelId":"glm-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k.","family":"tau3 Telecom","locator":"images/image4.png","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":98.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"tau3-telecom-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"tau3-telecom-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IHRhdTMgNCB0cmlhbHMsIEdQVDUuMiBsb3cgc2ltdWxhdG9yIGV4Y2VwdCBTaWVycmEgcmVwb3J0ZWQgY29tcGFyaXNvbnM7IEJyb3dzZUNvbXAgZGlzY2FyZC1hbGwgY29udGV4dCBhdDEwMGsu","id":"launch-mistral-1470","ingestRunId":"launch-mistral","modelId":"qwen3.5-397b-a17b","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k.","family":"tau3 Telecom","locator":"images/image4.png","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":97.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"tau3-airline-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"tau3-airline-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IHRhdTMgNCB0cmlhbHMsIEdQVDUuMiBsb3cgc2ltdWxhdG9yIGV4Y2VwdCBTaWVycmEgcmVwb3J0ZWQgY29tcGFyaXNvbnM7IEJyb3dzZUNvbXAgZGlzY2FyZC1hbGwgY29udGV4dCBhdDEwMGsu","id":"launch-mistral-1471","ingestRunId":"launch-mistral","modelId":"mistral-medium-3.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k.","family":"tau3 Airline","locator":"images/image4.png","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":72,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"tau3-airline-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"tau3-airline-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IHRhdTMgNCB0cmlhbHMsIEdQVDUuMiBsb3cgc2ltdWxhdG9yIGV4Y2VwdCBTaWVycmEgcmVwb3J0ZWQgY29tcGFyaXNvbnM7IEJyb3dzZUNvbXAgZGlzY2FyZC1hbGwgY29udGV4dCBhdDEwMGsu","id":"launch-mistral-1472","ingestRunId":"launch-mistral","modelId":"claude-sonnet-4-5-20250929","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k.","family":"tau3 Airline","locator":"images/image4.png","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":72,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"tau3-airline-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"tau3-airline-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IHRhdTMgNCB0cmlhbHMsIEdQVDUuMiBsb3cgc2ltdWxhdG9yIGV4Y2VwdCBTaWVycmEgcmVwb3J0ZWQgY29tcGFyaXNvbnM7IEJyb3dzZUNvbXAgZGlzY2FyZC1hbGwgY29udGV4dCBhdDEwMGsu","id":"launch-mistral-1473","ingestRunId":"launch-mistral","modelId":"claude-sonnet-4-6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k.","family":"tau3 Airline","locator":"images/image4.png","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":83,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"tau3-airline-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"tau3-airline-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IHRhdTMgNCB0cmlhbHMsIEdQVDUuMiBsb3cgc2ltdWxhdG9yIGV4Y2VwdCBTaWVycmEgcmVwb3J0ZWQgY29tcGFyaXNvbnM7IEJyb3dzZUNvbXAgZGlzY2FyZC1hbGwgY29udGV4dCBhdDEwMGsu","id":"launch-mistral-1474","ingestRunId":"launch-mistral","modelId":"glm-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k.","family":"tau3 Airline","locator":"images/image4.png","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":79.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"tau3-airline-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"tau3-airline-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IHRhdTMgNCB0cmlhbHMsIEdQVDUuMiBsb3cgc2ltdWxhdG9yIGV4Y2VwdCBTaWVycmEgcmVwb3J0ZWQgY29tcGFyaXNvbnM7IEJyb3dzZUNvbXAgZGlzY2FyZC1hbGwgY29udGV4dCBhdDEwMGsu","id":"launch-mistral-1475","ingestRunId":"launch-mistral","modelId":"qwen3.5-397b-a17b","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k.","family":"tau3 Airline","locator":"images/image4.png","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":81.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"tau3-retail-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"tau3-retail-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IHRhdTMgNCB0cmlhbHMsIEdQVDUuMiBsb3cgc2ltdWxhdG9yIGV4Y2VwdCBTaWVycmEgcmVwb3J0ZWQgY29tcGFyaXNvbnM7IEJyb3dzZUNvbXAgZGlzY2FyZC1hbGwgY29udGV4dCBhdDEwMGsu","id":"launch-mistral-1476","ingestRunId":"launch-mistral","modelId":"mistral-medium-3.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k.","family":"tau3 Retail","locator":"images/image4.png","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":76.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"tau3-retail-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"tau3-retail-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IHRhdTMgNCB0cmlhbHMsIEdQVDUuMiBsb3cgc2ltdWxhdG9yIGV4Y2VwdCBTaWVycmEgcmVwb3J0ZWQgY29tcGFyaXNvbnM7IEJyb3dzZUNvbXAgZGlzY2FyZC1hbGwgY29udGV4dCBhdDEwMGsu","id":"launch-mistral-1477","ingestRunId":"launch-mistral","modelId":"claude-sonnet-4-5-20250929","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k.","family":"tau3 Retail","locator":"images/image4.png","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":72.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"tau3-retail-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"tau3-retail-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IHRhdTMgNCB0cmlhbHMsIEdQVDUuMiBsb3cgc2ltdWxhdG9yIGV4Y2VwdCBTaWVycmEgcmVwb3J0ZWQgY29tcGFyaXNvbnM7IEJyb3dzZUNvbXAgZGlzY2FyZC1hbGwgY29udGV4dCBhdDEwMGsu","id":"launch-mistral-1478","ingestRunId":"launch-mistral","modelId":"claude-sonnet-4-6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k.","family":"tau3 Retail","locator":"images/image4.png","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":75.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"tau3-retail-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"tau3-retail-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IHRhdTMgNCB0cmlhbHMsIEdQVDUuMiBsb3cgc2ltdWxhdG9yIGV4Y2VwdCBTaWVycmEgcmVwb3J0ZWQgY29tcGFyaXNvbnM7IEJyb3dzZUNvbXAgZGlzY2FyZC1hbGwgY29udGV4dCBhdDEwMGsu","id":"launch-mistral-1479","ingestRunId":"launch-mistral","modelId":"glm-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k.","family":"tau3 Retail","locator":"images/image4.png","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":76.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"tau3-retail-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"tau3-retail-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IHRhdTMgNCB0cmlhbHMsIEdQVDUuMiBsb3cgc2ltdWxhdG9yIGV4Y2VwdCBTaWVycmEgcmVwb3J0ZWQgY29tcGFyaXNvbnM7IEJyb3dzZUNvbXAgZGlzY2FyZC1hbGwgY29udGV4dCBhdDEwMGsu","id":"launch-mistral-1480","ingestRunId":"launch-mistral","modelId":"qwen3.5-397b-a17b","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k.","family":"tau3 Retail","locator":"images/image4.png","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":84.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"tau3-banking-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"tau3-banking-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IHRhdTMgNCB0cmlhbHMsIEdQVDUuMiBsb3cgc2ltdWxhdG9yIGV4Y2VwdCBTaWVycmEgcmVwb3J0ZWQgY29tcGFyaXNvbnM7IEJyb3dzZUNvbXAgZGlzY2FyZC1hbGwgY29udGV4dCBhdDEwMGsu","id":"launch-mistral-1481","ingestRunId":"launch-mistral","modelId":"mistral-medium-3.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k.","family":"tau3 Banking","locator":"images/image4.png","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":13.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"tau3-banking-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"tau3-banking-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IHRhdTMgNCB0cmlhbHMsIEdQVDUuMiBsb3cgc2ltdWxhdG9yIGV4Y2VwdCBTaWVycmEgcmVwb3J0ZWQgY29tcGFyaXNvbnM7IEJyb3dzZUNvbXAgZGlzY2FyZC1hbGwgY29udGV4dCBhdDEwMGsu","id":"launch-mistral-1482","ingestRunId":"launch-mistral","modelId":"claude-sonnet-4-5-20250929","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k.","family":"tau3 Banking","locator":"images/image4.png","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":22.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"tau3-banking-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"tau3-banking-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IHRhdTMgNCB0cmlhbHMsIEdQVDUuMiBsb3cgc2ltdWxhdG9yIGV4Y2VwdCBTaWVycmEgcmVwb3J0ZWQgY29tcGFyaXNvbnM7IEJyb3dzZUNvbXAgZGlzY2FyZC1hbGwgY29udGV4dCBhdDEwMGsu","id":"launch-mistral-1483","ingestRunId":"launch-mistral","modelId":"claude-sonnet-4-6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k.","family":"tau3 Banking","locator":"images/image4.png","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":28.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"tau3-banking-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"tau3-banking-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IHRhdTMgNCB0cmlhbHMsIEdQVDUuMiBsb3cgc2ltdWxhdG9yIGV4Y2VwdCBTaWVycmEgcmVwb3J0ZWQgY29tcGFyaXNvbnM7IEJyb3dzZUNvbXAgZGlzY2FyZC1hbGwgY29udGV4dCBhdDEwMGsu","id":"launch-mistral-1484","ingestRunId":"launch-mistral","modelId":"glm-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k.","family":"tau3 Banking","locator":"images/image4.png","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":16.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"tau3-banking-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"tau3-banking-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IHRhdTMgNCB0cmlhbHMsIEdQVDUuMiBsb3cgc2ltdWxhdG9yIGV4Y2VwdCBTaWVycmEgcmVwb3J0ZWQgY29tcGFyaXNvbnM7IEJyb3dzZUNvbXAgZGlzY2FyZC1hbGwgY29udGV4dCBhdDEwMGsu","id":"launch-mistral-1485","ingestRunId":"launch-mistral","modelId":"qwen3.5-397b-a17b","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k.","family":"tau3 Banking","locator":"images/image4.png","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":9.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"browsecomp-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"browsecomp-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IHRhdTMgNCB0cmlhbHMsIEdQVDUuMiBsb3cgc2ltdWxhdG9yIGV4Y2VwdCBTaWVycmEgcmVwb3J0ZWQgY29tcGFyaXNvbnM7IEJyb3dzZUNvbXAgZGlzY2FyZC1hbGwgY29udGV4dCBhdDEwMGsu","id":"launch-mistral-1486","ingestRunId":"launch-mistral","modelId":"mistral-medium-3.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k.","family":"BrowseComp","locator":"images/image4.png","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":48.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"browsecomp-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"browsecomp-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IHRhdTMgNCB0cmlhbHMsIEdQVDUuMiBsb3cgc2ltdWxhdG9yIGV4Y2VwdCBTaWVycmEgcmVwb3J0ZWQgY29tcGFyaXNvbnM7IEJyb3dzZUNvbXAgZGlzY2FyZC1hbGwgY29udGV4dCBhdDEwMGsu","id":"launch-mistral-1487","ingestRunId":"launch-mistral","modelId":"claude-sonnet-4-5-20250929","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k.","family":"BrowseComp","locator":"images/image4.png","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":43.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"browsecomp-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"browsecomp-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IHRhdTMgNCB0cmlhbHMsIEdQVDUuMiBsb3cgc2ltdWxhdG9yIGV4Y2VwdCBTaWVycmEgcmVwb3J0ZWQgY29tcGFyaXNvbnM7IEJyb3dzZUNvbXAgZGlzY2FyZC1hbGwgY29udGV4dCBhdDEwMGsu","id":"launch-mistral-1488","ingestRunId":"launch-mistral","modelId":"claude-sonnet-4-6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k.","family":"BrowseComp","locator":"images/image4.png","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":74.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"browsecomp-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"browsecomp-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IHRhdTMgNCB0cmlhbHMsIEdQVDUuMiBsb3cgc2ltdWxhdG9yIGV4Y2VwdCBTaWVycmEgcmVwb3J0ZWQgY29tcGFyaXNvbnM7IEJyb3dzZUNvbXAgZGlzY2FyZC1hbGwgY29udGV4dCBhdDEwMGsu","id":"launch-mistral-1489","ingestRunId":"launch-mistral","modelId":"glm-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k.","family":"BrowseComp","locator":"images/image4.png","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":79.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"browsecomp-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"browsecomp-source-release-snapshot-version-not-specified:mistral:TWF4aW11bSByZWFzb25pbmc7IHRhdTMgNCB0cmlhbHMsIEdQVDUuMiBsb3cgc2ltdWxhdG9yIGV4Y2VwdCBTaWVycmEgcmVwb3J0ZWQgY29tcGFyaXNvbnM7IEJyb3dzZUNvbXAgZGlzY2FyZC1hbGwgY29udGV4dCBhdDEwMGsu","id":"launch-mistral-1490","ingestRunId":"launch-mistral","modelId":"qwen3.5-397b-a17b","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k.","family":"BrowseComp","locator":"images/image4.png","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":78.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"tau3-telecom-source-release-snapshot-version-as-labeled","evidenceKind":"lab_self_report","harnessId":"tau3-telecom-source-release-snapshot-version-as-labeled:mistral:TWF4aW11bSByZWFzb25pbmc7IE1pc3RyYWwgc2FtZS1sYWIgY29tcGFyaXNvbjsgdGF1MyA0dHJpYWxzIEdQVDUuMmxvdyB1c2VyIHNpbXVsYXRvcjsgQnJvd3NlQ29tcCBjb250ZXh0LWRpc2NhcmQgc2V0dXAgaW4gY2FyZC4","id":"launch-mistral-1950","ingestRunId":"launch-mistral","modelId":"magistral-medium-2509","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reasoning; Mistral same-lab comparison; tau3 4trials GPT5.2low user simulator; BrowseComp context-discard setup in card.","family":"tau3 Telecom","locator":"images/image2.png","metric":"%","notes":"First-party chart numeric label, visually verified.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":60.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"tau3-airline-source-release-snapshot-version-as-labeled","evidenceKind":"lab_self_report","harnessId":"tau3-airline-source-release-snapshot-version-as-labeled:mistral:TWF4aW11bSByZWFzb25pbmc7IE1pc3RyYWwgc2FtZS1sYWIgY29tcGFyaXNvbjsgdGF1MyA0dHJpYWxzIEdQVDUuMmxvdyB1c2VyIHNpbXVsYXRvcjsgQnJvd3NlQ29tcCBjb250ZXh0LWRpc2NhcmQgc2V0dXAgaW4gY2FyZC4","id":"launch-mistral-1951","ingestRunId":"launch-mistral","modelId":"magistral-medium-2509","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reasoning; Mistral same-lab comparison; tau3 4trials GPT5.2low user simulator; BrowseComp context-discard setup in card.","family":"tau3 Airline","locator":"images/image2.png","metric":"%","notes":"First-party chart numeric label, visually verified.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":53.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"tau3-retail-source-release-snapshot-version-as-labeled","evidenceKind":"lab_self_report","harnessId":"tau3-retail-source-release-snapshot-version-as-labeled:mistral:TWF4aW11bSByZWFzb25pbmc7IE1pc3RyYWwgc2FtZS1sYWIgY29tcGFyaXNvbjsgdGF1MyA0dHJpYWxzIEdQVDUuMmxvdyB1c2VyIHNpbXVsYXRvcjsgQnJvd3NlQ29tcCBjb250ZXh0LWRpc2NhcmQgc2V0dXAgaW4gY2FyZC4","id":"launch-mistral-1952","ingestRunId":"launch-mistral","modelId":"magistral-medium-2509","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reasoning; Mistral same-lab comparison; tau3 4trials GPT5.2low user simulator; BrowseComp context-discard setup in card.","family":"tau3 Retail","locator":"images/image2.png","metric":"%","notes":"First-party chart numeric label, visually verified.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":70.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"tau3-banking-source-release-snapshot-version-as-labeled","evidenceKind":"lab_self_report","harnessId":"tau3-banking-source-release-snapshot-version-as-labeled:mistral:TWF4aW11bSByZWFzb25pbmc7IE1pc3RyYWwgc2FtZS1sYWIgY29tcGFyaXNvbjsgdGF1MyA0dHJpYWxzIEdQVDUuMmxvdyB1c2VyIHNpbXVsYXRvcjsgQnJvd3NlQ29tcCBjb250ZXh0LWRpc2NhcmQgc2V0dXAgaW4gY2FyZC4","id":"launch-mistral-1953","ingestRunId":"launch-mistral","modelId":"magistral-medium-2509","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reasoning; Mistral same-lab comparison; tau3 4trials GPT5.2low user simulator; BrowseComp context-discard setup in card.","family":"tau3 Banking","locator":"images/image2.png","metric":"%","notes":"First-party chart numeric label, visually verified.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":7.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"browsecomp-source-release-snapshot-version-as-labeled","evidenceKind":"lab_self_report","harnessId":"browsecomp-source-release-snapshot-version-as-labeled:mistral:TWF4aW11bSByZWFzb25pbmc7IE1pc3RyYWwgc2FtZS1sYWIgY29tcGFyaXNvbjsgdGF1MyA0dHJpYWxzIEdQVDUuMmxvdyB1c2VyIHNpbXVsYXRvcjsgQnJvd3NlQ29tcCBjb250ZXh0LWRpc2NhcmQgc2V0dXAgaW4gY2FyZC4","id":"launch-mistral-1954","ingestRunId":"launch-mistral","modelId":"magistral-medium-2509","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reasoning; Mistral same-lab comparison; tau3 4trials GPT5.2low user simulator; BrowseComp context-discard setup in card.","family":"BrowseComp","locator":"images/image2.png","metric":"%","notes":"First-party chart numeric label, visually verified.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":10,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"tau3-telecom-source-release-snapshot-version-as-labeled","evidenceKind":"lab_self_report","harnessId":"tau3-telecom-source-release-snapshot-version-as-labeled:mistral:TWF4aW11bSByZWFzb25pbmc7IE1pc3RyYWwgc2FtZS1sYWIgY29tcGFyaXNvbjsgdGF1MyA0dHJpYWxzIEdQVDUuMmxvdyB1c2VyIHNpbXVsYXRvcjsgQnJvd3NlQ29tcCBjb250ZXh0LWRpc2NhcmQgc2V0dXAgaW4gY2FyZC4","id":"launch-mistral-1955","ingestRunId":"launch-mistral","modelId":"mistral-small-4","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reasoning; Mistral same-lab comparison; tau3 4trials GPT5.2low user simulator; BrowseComp context-discard setup in card.","family":"tau3 Telecom","locator":"images/image2.png","metric":"%","notes":"First-party chart numeric label, visually verified.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":47.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"tau3-airline-source-release-snapshot-version-as-labeled","evidenceKind":"lab_self_report","harnessId":"tau3-airline-source-release-snapshot-version-as-labeled:mistral:TWF4aW11bSByZWFzb25pbmc7IE1pc3RyYWwgc2FtZS1sYWIgY29tcGFyaXNvbjsgdGF1MyA0dHJpYWxzIEdQVDUuMmxvdyB1c2VyIHNpbXVsYXRvcjsgQnJvd3NlQ29tcCBjb250ZXh0LWRpc2NhcmQgc2V0dXAgaW4gY2FyZC4","id":"launch-mistral-1956","ingestRunId":"launch-mistral","modelId":"mistral-small-4","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reasoning; Mistral same-lab comparison; tau3 4trials GPT5.2low user simulator; BrowseComp context-discard setup in card.","family":"tau3 Airline","locator":"images/image2.png","metric":"%","notes":"First-party chart numeric label, visually verified.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":38.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"tau3-retail-source-release-snapshot-version-as-labeled","evidenceKind":"lab_self_report","harnessId":"tau3-retail-source-release-snapshot-version-as-labeled:mistral:TWF4aW11bSByZWFzb25pbmc7IE1pc3RyYWwgc2FtZS1sYWIgY29tcGFyaXNvbjsgdGF1MyA0dHJpYWxzIEdQVDUuMmxvdyB1c2VyIHNpbXVsYXRvcjsgQnJvd3NlQ29tcCBjb250ZXh0LWRpc2NhcmQgc2V0dXAgaW4gY2FyZC4","id":"launch-mistral-1957","ingestRunId":"launch-mistral","modelId":"mistral-small-4","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reasoning; Mistral same-lab comparison; tau3 4trials GPT5.2low user simulator; BrowseComp context-discard setup in card.","family":"tau3 Retail","locator":"images/image2.png","metric":"%","notes":"First-party chart numeric label, visually verified.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":67.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"tau3-banking-source-release-snapshot-version-as-labeled","evidenceKind":"lab_self_report","harnessId":"tau3-banking-source-release-snapshot-version-as-labeled:mistral:TWF4aW11bSByZWFzb25pbmc7IE1pc3RyYWwgc2FtZS1sYWIgY29tcGFyaXNvbjsgdGF1MyA0dHJpYWxzIEdQVDUuMmxvdyB1c2VyIHNpbXVsYXRvcjsgQnJvd3NlQ29tcCBjb250ZXh0LWRpc2NhcmQgc2V0dXAgaW4gY2FyZC4","id":"launch-mistral-1958","ingestRunId":"launch-mistral","modelId":"mistral-small-4","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reasoning; Mistral same-lab comparison; tau3 4trials GPT5.2low user simulator; BrowseComp context-discard setup in card.","family":"tau3 Banking","locator":"images/image2.png","metric":"%","notes":"First-party chart numeric label, visually verified.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"browsecomp-source-release-snapshot-version-as-labeled","evidenceKind":"lab_self_report","harnessId":"browsecomp-source-release-snapshot-version-as-labeled:mistral:TWF4aW11bSByZWFzb25pbmc7IE1pc3RyYWwgc2FtZS1sYWIgY29tcGFyaXNvbjsgdGF1MyA0dHJpYWxzIEdQVDUuMmxvdyB1c2VyIHNpbXVsYXRvcjsgQnJvd3NlQ29tcCBjb250ZXh0LWRpc2NhcmQgc2V0dXAgaW4gY2FyZC4","id":"launch-mistral-1959","ingestRunId":"launch-mistral","modelId":"mistral-small-4","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reasoning; Mistral same-lab comparison; tau3 4trials GPT5.2low user simulator; BrowseComp context-discard setup in card.","family":"BrowseComp","locator":"images/image2.png","metric":"%","notes":"First-party chart numeric label, visually verified.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":21.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"tau3-telecom-source-release-snapshot-version-as-labeled","evidenceKind":"lab_self_report","harnessId":"tau3-telecom-source-release-snapshot-version-as-labeled:mistral:TWF4aW11bSByZWFzb25pbmc7IE1pc3RyYWwgc2FtZS1sYWIgY29tcGFyaXNvbjsgdGF1MyA0dHJpYWxzIEdQVDUuMmxvdyB1c2VyIHNpbXVsYXRvcjsgQnJvd3NlQ29tcCBjb250ZXh0LWRpc2NhcmQgc2V0dXAgaW4gY2FyZC4","id":"launch-mistral-1960","ingestRunId":"launch-mistral","modelId":"mistral-medium-2508","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reasoning; Mistral same-lab comparison; tau3 4trials GPT5.2low user simulator; BrowseComp context-discard setup in card.","family":"tau3 Telecom","locator":"images/image2.png","metric":"%","notes":"First-party chart numeric label, visually verified.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":46.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"tau3-airline-source-release-snapshot-version-as-labeled","evidenceKind":"lab_self_report","harnessId":"tau3-airline-source-release-snapshot-version-as-labeled:mistral:TWF4aW11bSByZWFzb25pbmc7IE1pc3RyYWwgc2FtZS1sYWIgY29tcGFyaXNvbjsgdGF1MyA0dHJpYWxzIEdQVDUuMmxvdyB1c2VyIHNpbXVsYXRvcjsgQnJvd3NlQ29tcCBjb250ZXh0LWRpc2NhcmQgc2V0dXAgaW4gY2FyZC4","id":"launch-mistral-1961","ingestRunId":"launch-mistral","modelId":"mistral-medium-2508","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reasoning; Mistral same-lab comparison; tau3 4trials GPT5.2low user simulator; BrowseComp context-discard setup in card.","family":"tau3 Airline","locator":"images/image2.png","metric":"%","notes":"First-party chart numeric label, visually verified.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":41.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"tau3-retail-source-release-snapshot-version-as-labeled","evidenceKind":"lab_self_report","harnessId":"tau3-retail-source-release-snapshot-version-as-labeled:mistral:TWF4aW11bSByZWFzb25pbmc7IE1pc3RyYWwgc2FtZS1sYWIgY29tcGFyaXNvbjsgdGF1MyA0dHJpYWxzIEdQVDUuMmxvdyB1c2VyIHNpbXVsYXRvcjsgQnJvd3NlQ29tcCBjb250ZXh0LWRpc2NhcmQgc2V0dXAgaW4gY2FyZC4","id":"launch-mistral-1962","ingestRunId":"launch-mistral","modelId":"mistral-medium-2508","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reasoning; Mistral same-lab comparison; tau3 4trials GPT5.2low user simulator; BrowseComp context-discard setup in card.","family":"tau3 Retail","locator":"images/image2.png","metric":"%","notes":"First-party chart numeric label, visually verified.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":64.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"tau3-banking-source-release-snapshot-version-as-labeled","evidenceKind":"lab_self_report","harnessId":"tau3-banking-source-release-snapshot-version-as-labeled:mistral:TWF4aW11bSByZWFzb25pbmc7IE1pc3RyYWwgc2FtZS1sYWIgY29tcGFyaXNvbjsgdGF1MyA0dHJpYWxzIEdQVDUuMmxvdyB1c2VyIHNpbXVsYXRvcjsgQnJvd3NlQ29tcCBjb250ZXh0LWRpc2NhcmQgc2V0dXAgaW4gY2FyZC4","id":"launch-mistral-1963","ingestRunId":"launch-mistral","modelId":"mistral-medium-2508","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reasoning; Mistral same-lab comparison; tau3 4trials GPT5.2low user simulator; BrowseComp context-discard setup in card.","family":"tau3 Banking","locator":"images/image2.png","metric":"%","notes":"First-party chart numeric label, visually verified.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":5.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"browsecomp-source-release-snapshot-version-as-labeled","evidenceKind":"lab_self_report","harnessId":"browsecomp-source-release-snapshot-version-as-labeled:mistral:TWF4aW11bSByZWFzb25pbmc7IE1pc3RyYWwgc2FtZS1sYWIgY29tcGFyaXNvbjsgdGF1MyA0dHJpYWxzIEdQVDUuMmxvdyB1c2VyIHNpbXVsYXRvcjsgQnJvd3NlQ29tcCBjb250ZXh0LWRpc2NhcmQgc2V0dXAgaW4gY2FyZC4","id":"launch-mistral-1964","ingestRunId":"launch-mistral","modelId":"mistral-medium-2508","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reasoning; Mistral same-lab comparison; tau3 4trials GPT5.2low user simulator; BrowseComp context-discard setup in card.","family":"BrowseComp","locator":"images/image2.png","metric":"%","notes":"First-party chart numeric label, visually verified.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":7.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"swe-bench-verified-source-release-snapshot-version-as-labeled","evidenceKind":"lab_self_report","harnessId":"swe-bench-verified-source-release-snapshot-version-as-labeled:mistral:TWlzdHJhbCBwcmV2aW91cyBjb2RpbmcgbW9kZWwgY29tcGFyaXNvbiwgcHVibGlzaGVkIG1vZGVsLWNhcmQgaGFybmVzcyBzZXR0aW5ncy4","id":"launch-mistral-1965","ingestRunId":"launch-mistral","modelId":"devstral-2512","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Mistral previous coding model comparison, published model-card harness settings.","family":"SWE-bench Verified","locator":"images/image3.png","metric":"%","notes":"First-party chart numeric label, visually verified.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":72.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"swe-bench-verified-source-release-snapshot-version-as-labeled","evidenceKind":"lab_self_report","harnessId":"swe-bench-verified-source-release-snapshot-version-as-labeled:mistral:TWlzdHJhbCBwcmV2aW91cyBjb2RpbmcgbW9kZWwgY29tcGFyaXNvbiwgcHVibGlzaGVkIG1vZGVsLWNhcmQgaGFybmVzcyBzZXR0aW5ncy4","id":"launch-mistral-1966","ingestRunId":"launch-mistral","modelId":"labs-devstral-small-2512","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Mistral previous coding model comparison, published model-card harness settings.","family":"SWE-bench Verified","locator":"images/image3.png","metric":"%","notes":"First-party chart numeric label, visually verified.","sourceId":"mistral","sourceTitle":"Mistral Medium 3.5 model card performance charts"},"score":68,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B"},{"benchmarkId":"pinchbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"pinchbench-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1000","ingestRunId":"launch-nvidia","modelId":"minimax-m2.7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"PinchBench","locator":"Performance table: PinchBench","metric":"%","notes":"First-party reported result. Original cell: 77.6","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":77.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"pinchbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"pinchbench-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1001","ingestRunId":"launch-nvidia","modelId":"glm-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"PinchBench","locator":"Performance table: PinchBench","metric":"%","notes":"First-party reported result. Original cell: 81.2","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":81.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"pinchbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"pinchbench-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1002","ingestRunId":"launch-nvidia","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"PinchBench","locator":"Performance table: PinchBench","metric":"%","notes":"First-party reported result. Original cell: 90.2","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":90.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"pinchbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"pinchbench-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1003","ingestRunId":"launch-nvidia","modelId":"qwen3.5-397b-a17b","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"PinchBench","locator":"Performance table: PinchBench","metric":"%","notes":"First-party reported result. Original cell: 86.6","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":86.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"taubench-v3-airline-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"taubench-v3-airline-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1004","ingestRunId":"launch-nvidia","modelId":"nemotron-3-ultra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"TauBench V3 Airline","locator":"Performance table: Airline","metric":"%","notes":"First-party reported result. Original cell: 81.5","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":81.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"taubench-v3-airline-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"taubench-v3-airline-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1005","ingestRunId":"launch-nvidia","modelId":"minimax-m2.7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"TauBench V3 Airline","locator":"Performance table: Airline","metric":"%","notes":"First-party reported result. Original cell: 75.3","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":75.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"taubench-v3-airline-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"taubench-v3-airline-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1006","ingestRunId":"launch-nvidia","modelId":"glm-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"TauBench V3 Airline","locator":"Performance table: Airline","metric":"%","notes":"First-party reported result. Original cell: 85.0","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":85,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"taubench-v3-airline-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"taubench-v3-airline-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1007","ingestRunId":"launch-nvidia","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"TauBench V3 Airline","locator":"Performance table: Airline","metric":"%","notes":"First-party reported result. Original cell: 85.8","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":85.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"taubench-v3-airline-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"taubench-v3-airline-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1008","ingestRunId":"launch-nvidia","modelId":"qwen3.5-397b-a17b","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"TauBench V3 Airline","locator":"Performance table: Airline","metric":"%","notes":"First-party reported result. Original cell: 76.5","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":76.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"taubench-v3-retail-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"taubench-v3-retail-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1009","ingestRunId":"launch-nvidia","modelId":"nemotron-3-ultra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"TauBench V3 Retail","locator":"Performance table: Retail","metric":"%","notes":"First-party reported result. Original cell: 86.4","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":86.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"taubench-v3-retail-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"taubench-v3-retail-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1010","ingestRunId":"launch-nvidia","modelId":"minimax-m2.7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"TauBench V3 Retail","locator":"Performance table: Retail","metric":"%","notes":"First-party reported result. Original cell: 84.9","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":84.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"taubench-v3-retail-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"taubench-v3-retail-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1011","ingestRunId":"launch-nvidia","modelId":"glm-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"TauBench V3 Retail","locator":"Performance table: Retail","metric":"%","notes":"First-party reported result. Original cell: 84.1","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":84.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"taubench-v3-retail-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"taubench-v3-retail-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1012","ingestRunId":"launch-nvidia","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"TauBench V3 Retail","locator":"Performance table: Retail","metric":"%","notes":"First-party reported result. Original cell: 82.9","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":82.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"taubench-v3-retail-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"taubench-v3-retail-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1013","ingestRunId":"launch-nvidia","modelId":"qwen3.5-397b-a17b","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"TauBench V3 Retail","locator":"Performance table: Retail","metric":"%","notes":"First-party reported result. Original cell: 88.5","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":88.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"taubench-v3-telecom-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"taubench-v3-telecom-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1014","ingestRunId":"launch-nvidia","modelId":"nemotron-3-ultra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"TauBench V3 Telecom","locator":"Performance table: Telecom","metric":"%","notes":"First-party reported result. Original cell: 92.9","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":92.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"taubench-v3-telecom-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"taubench-v3-telecom-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1015","ingestRunId":"launch-nvidia","modelId":"minimax-m2.7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"TauBench V3 Telecom","locator":"Performance table: Telecom","metric":"%","notes":"First-party reported result. Original cell: 89.6","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":89.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"taubench-v3-telecom-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"taubench-v3-telecom-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1016","ingestRunId":"launch-nvidia","modelId":"glm-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"TauBench V3 Telecom","locator":"Performance table: Telecom","metric":"%","notes":"First-party reported result. Original cell: 96.9","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":96.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"taubench-v3-telecom-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"taubench-v3-telecom-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1017","ingestRunId":"launch-nvidia","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"TauBench V3 Telecom","locator":"Performance table: Telecom","metric":"%","notes":"First-party reported result. Original cell: 97.8","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":97.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"taubench-v3-telecom-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"taubench-v3-telecom-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1018","ingestRunId":"launch-nvidia","modelId":"qwen3.5-397b-a17b","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"TauBench V3 Telecom","locator":"Performance table: Telecom","metric":"%","notes":"First-party reported result. Original cell: 98.0","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":98,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"taubench-v3-banking-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"taubench-v3-banking-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1019","ingestRunId":"launch-nvidia","modelId":"nemotron-3-ultra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"TauBench V3 Banking","locator":"Performance table: Banking","metric":"%","notes":"First-party reported result. Original cell: 22.6","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":22.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"taubench-v3-banking-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"taubench-v3-banking-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1020","ingestRunId":"launch-nvidia","modelId":"minimax-m2.7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"TauBench V3 Banking","locator":"Performance table: Banking","metric":"%","notes":"First-party reported result. Original cell: 14.6","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":14.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"taubench-v3-banking-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"taubench-v3-banking-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1021","ingestRunId":"launch-nvidia","modelId":"glm-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"TauBench V3 Banking","locator":"Performance table: Banking","metric":"%","notes":"First-party reported result. Original cell: 12.8","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":12.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"taubench-v3-banking-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"taubench-v3-banking-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1022","ingestRunId":"launch-nvidia","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"TauBench V3 Banking","locator":"Performance table: Banking","metric":"%","notes":"First-party reported result. Original cell: 23.1","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":23.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"taubench-v3-banking-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"taubench-v3-banking-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1023","ingestRunId":"launch-nvidia","modelId":"qwen3.5-397b-a17b","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"TauBench V3 Banking","locator":"Performance table: Banking","metric":"%","notes":"First-party reported result. Original cell: 20.9","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":20.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"taubench-v3-average-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"taubench-v3-average-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1024","ingestRunId":"launch-nvidia","modelId":"nemotron-3-ultra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"TauBench V3 Average","locator":"Performance table: Average","metric":"%","notes":"First-party reported result. Original cell: 70.9","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":70.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"taubench-v3-average-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"taubench-v3-average-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1025","ingestRunId":"launch-nvidia","modelId":"minimax-m2.7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"TauBench V3 Average","locator":"Performance table: Average","metric":"%","notes":"First-party reported result. Original cell: 66.1","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":66.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"taubench-v3-average-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"taubench-v3-average-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1026","ingestRunId":"launch-nvidia","modelId":"glm-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"TauBench V3 Average","locator":"Performance table: Average","metric":"%","notes":"First-party reported result. Original cell: 69.7","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":69.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"taubench-v3-average-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"taubench-v3-average-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1027","ingestRunId":"launch-nvidia","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"TauBench V3 Average","locator":"Performance table: Average","metric":"%","notes":"First-party reported result. Original cell: 72.4","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":72.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"taubench-v3-average-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"taubench-v3-average-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1028","ingestRunId":"launch-nvidia","modelId":"qwen3.5-397b-a17b","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"TauBench V3 Average","locator":"Performance table: Average","metric":"%","notes":"First-party reported result. Original cell: 71.0","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":71,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"browsecomp-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"browsecomp-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1029","ingestRunId":"launch-nvidia","modelId":"nemotron-3-ultra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"BrowseComp","locator":"Performance table: BrowseComp","metric":"%","notes":"First-party reported result. Original cell: 44.4","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":44.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"browsecomp-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"browsecomp-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1030","ingestRunId":"launch-nvidia","modelId":"minimax-m2.7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"BrowseComp","locator":"Performance table: BrowseComp","metric":"%","notes":"First-party reported result. Original cell: 54.1","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":54.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"browsecomp-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"browsecomp-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1031","ingestRunId":"launch-nvidia","modelId":"glm-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"BrowseComp","locator":"Performance table: BrowseComp","metric":"%","notes":"First-party reported result. Original cell: 59.4","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":59.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"browsecomp-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"browsecomp-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1032","ingestRunId":"launch-nvidia","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"BrowseComp","locator":"Performance table: BrowseComp","metric":"%","notes":"First-party reported result. Original cell: 61.3","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":61.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"browsecomp-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"browsecomp-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1033","ingestRunId":"launch-nvidia","modelId":"qwen3.5-397b-a17b","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"BrowseComp","locator":"Performance table: BrowseComp","metric":"%","notes":"First-party reported result. Original cell: 40.5","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":40.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"vals-ai-financial-agent-1-1-without-web-search-1-1","evidenceKind":"lab_self_report","harnessId":"vals-ai-financial-agent-1-1-without-web-search-1-1:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1034","ingestRunId":"launch-nvidia","modelId":"nemotron-3-ultra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"Vals.ai Financial Agent 1.1 without web search","locator":"Performance table: without web search","metric":"%","notes":"First-party reported result. Original cell: 60.1","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":60.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"vals-ai-financial-agent-1-1-without-web-search-1-1","evidenceKind":"lab_self_report","harnessId":"vals-ai-financial-agent-1-1-without-web-search-1-1:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1035","ingestRunId":"launch-nvidia","modelId":"minimax-m2.7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"Vals.ai Financial Agent 1.1 without web search","locator":"Performance table: without web search","metric":"%","notes":"First-party reported result. Original cell: 51.3","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":51.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"vals-ai-financial-agent-1-1-without-web-search-1-1","evidenceKind":"lab_self_report","harnessId":"vals-ai-financial-agent-1-1-without-web-search-1-1:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1036","ingestRunId":"launch-nvidia","modelId":"glm-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"Vals.ai Financial Agent 1.1 without web search","locator":"Performance table: without web search","metric":"%","notes":"First-party reported result. Original cell: 60.2","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":60.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"vals-ai-financial-agent-1-1-without-web-search-1-1","evidenceKind":"lab_self_report","harnessId":"vals-ai-financial-agent-1-1-without-web-search-1-1:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1037","ingestRunId":"launch-nvidia","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"Vals.ai Financial Agent 1.1 without web search","locator":"Performance table: without web search","metric":"%","notes":"First-party reported result. Original cell: 54.0","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":54,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"vals-ai-financial-agent-1-1-without-web-search-1-1","evidenceKind":"lab_self_report","harnessId":"vals-ai-financial-agent-1-1-without-web-search-1-1:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1038","ingestRunId":"launch-nvidia","modelId":"qwen3.5-397b-a17b","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"Vals.ai Financial Agent 1.1 without web search","locator":"Performance table: without web search","metric":"%","notes":"First-party reported result. Original cell: 61.3","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":61.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"vals-ai-financial-agent-1-1-with-web-search-1-1","evidenceKind":"lab_self_report","harnessId":"vals-ai-financial-agent-1-1-with-web-search-1-1:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1039","ingestRunId":"launch-nvidia","modelId":"nemotron-3-ultra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"Vals.ai Financial Agent 1.1 with web search","locator":"Performance table: with web search","metric":"%","notes":"First-party reported result. Original cell: 53.7","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":53.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"vals-ai-financial-agent-1-1-with-web-search-1-1","evidenceKind":"lab_self_report","harnessId":"vals-ai-financial-agent-1-1-with-web-search-1-1:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1040","ingestRunId":"launch-nvidia","modelId":"minimax-m2.7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"Vals.ai Financial Agent 1.1 with web search","locator":"Performance table: with web search","metric":"%","notes":"First-party reported result. Original cell: 50.5","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":50.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"vals-ai-financial-agent-1-1-with-web-search-1-1","evidenceKind":"lab_self_report","harnessId":"vals-ai-financial-agent-1-1-with-web-search-1-1:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1041","ingestRunId":"launch-nvidia","modelId":"glm-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"Vals.ai Financial Agent 1.1 with web search","locator":"Performance table: with web search","metric":"%","notes":"First-party reported result. Original cell: 60.7","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":60.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"vals-ai-financial-agent-1-1-with-web-search-1-1","evidenceKind":"lab_self_report","harnessId":"vals-ai-financial-agent-1-1-with-web-search-1-1:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1042","ingestRunId":"launch-nvidia","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"Vals.ai Financial Agent 1.1 with web search","locator":"Performance table: with web search","metric":"%","notes":"First-party reported result. Original cell: 58.8","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":58.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"vals-ai-financial-agent-1-1-with-web-search-1-1","evidenceKind":"lab_self_report","harnessId":"vals-ai-financial-agent-1-1-with-web-search-1-1:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1043","ingestRunId":"launch-nvidia","modelId":"qwen3.5-397b-a17b","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"Vals.ai Financial Agent 1.1 with web search","locator":"Performance table: with web search","metric":"%","notes":"First-party reported result. Original cell: 59.0","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":59,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"ioi-2025-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"ioi-2025-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1044","ingestRunId":"launch-nvidia","modelId":"nemotron-3-ultra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"IOI 2025","locator":"Performance table: IOI 2025","metric":"points","notes":"First-party reported result. Original cell: 570.0","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":570,"scoreUnit":"index","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"ioi-2025-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"ioi-2025-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1045","ingestRunId":"launch-nvidia","modelId":"glm-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"IOI 2025","locator":"Performance table: IOI 2025","metric":"points","notes":"First-party reported result. Original cell: 456.5","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":456.5,"scoreUnit":"index","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"ioi-2025-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"ioi-2025-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1046","ingestRunId":"launch-nvidia","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"IOI 2025","locator":"Performance table: IOI 2025","metric":"points","notes":"First-party reported result. Original cell: 585.0","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":585,"scoreUnit":"index","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"ioi-2025-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"ioi-2025-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1047","ingestRunId":"launch-nvidia","modelId":"qwen3.5-397b-a17b","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"IOI 2025","locator":"Performance table: IOI 2025","metric":"points","notes":"First-party reported result. Original cell: 441.3","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":441.3,"scoreUnit":"index","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"livecodebench-v6-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"livecodebench-v6-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1048","ingestRunId":"launch-nvidia","modelId":"nemotron-3-ultra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"LiveCodeBench","locator":"Performance table: LiveCodeBench (v6)","metric":"%","notes":"First-party reported result. Original cell: 89.0","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":89,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"livecodebench-v6-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"livecodebench-v6-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1049","ingestRunId":"launch-nvidia","modelId":"minimax-m2.7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"LiveCodeBench","locator":"Performance table: LiveCodeBench (v6)","metric":"%","notes":"First-party reported result. Original cell: 77.2","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":77.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"livecodebench-v6-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"livecodebench-v6-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1050","ingestRunId":"launch-nvidia","modelId":"glm-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"LiveCodeBench","locator":"Performance table: LiveCodeBench (v6)","metric":"%","notes":"First-party reported result. Original cell: 85.7","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":85.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"livecodebench-v6-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"livecodebench-v6-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1051","ingestRunId":"launch-nvidia","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"LiveCodeBench","locator":"Performance table: LiveCodeBench (v6)","metric":"%","notes":"First-party reported result. Original cell: 90.2","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":90.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"livecodebench-v6-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"livecodebench-v6-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1052","ingestRunId":"launch-nvidia","modelId":"qwen3.5-397b-a17b","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"LiveCodeBench","locator":"Performance table: LiveCodeBench (v6)","metric":"%","notes":"First-party reported result. Original cell: 79.3","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":79.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"imoanswerbench-no-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"imoanswerbench-no-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1053","ingestRunId":"launch-nvidia","modelId":"nemotron-3-ultra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"IMOAnswerBench (no tools)","locator":"Performance table: IMOAnswerBench (no tools)","metric":"%","notes":"First-party reported result. Original cell: 88.6","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":88.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"imoanswerbench-no-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"imoanswerbench-no-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1054","ingestRunId":"launch-nvidia","modelId":"minimax-m2.7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"IMOAnswerBench (no tools)","locator":"Performance table: IMOAnswerBench (no tools)","metric":"%","notes":"First-party reported result. Original cell: 68.3","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":68.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"imoanswerbench-no-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"imoanswerbench-no-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1055","ingestRunId":"launch-nvidia","modelId":"glm-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"IMOAnswerBench (no tools)","locator":"Performance table: IMOAnswerBench (no tools)","metric":"%","notes":"First-party reported result. Original cell: 86.8","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":86.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"imoanswerbench-no-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"imoanswerbench-no-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1056","ingestRunId":"launch-nvidia","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"IMOAnswerBench (no tools)","locator":"Performance table: IMOAnswerBench (no tools)","metric":"%","notes":"First-party reported result. Original cell: 91.1","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":91.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"imoanswerbench-no-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"imoanswerbench-no-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1057","ingestRunId":"launch-nvidia","modelId":"qwen3.5-397b-a17b","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"IMOAnswerBench (no tools)","locator":"Performance table: IMOAnswerBench (no tools)","metric":"%","notes":"First-party reported result. Original cell: 83.1","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":83.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"imoanswerbench-with-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"imoanswerbench-with-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1058","ingestRunId":"launch-nvidia","modelId":"nemotron-3-ultra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"IMOAnswerBench (with tools)","locator":"Performance table: IMOAnswerBench (with tools)","metric":"%","notes":"First-party reported result. Original cell: 92.3","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":92.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"imoanswerbench-with-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"imoanswerbench-with-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1059","ingestRunId":"launch-nvidia","modelId":"minimax-m2.7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"IMOAnswerBench (with tools)","locator":"Performance table: IMOAnswerBench (with tools)","metric":"%","notes":"First-party reported result. Original cell: 75.1","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":75.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"imoanswerbench-with-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"imoanswerbench-with-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1060","ingestRunId":"launch-nvidia","modelId":"glm-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"IMOAnswerBench (with tools)","locator":"Performance table: IMOAnswerBench (with tools)","metric":"%","notes":"First-party reported result. Original cell: 91.1","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":91.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"imoanswerbench-with-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"imoanswerbench-with-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1061","ingestRunId":"launch-nvidia","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"IMOAnswerBench (with tools)","locator":"Performance table: IMOAnswerBench (with tools)","metric":"%","notes":"First-party reported result. Original cell: 93.71","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":93.71,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"imoanswerbench-with-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"imoanswerbench-with-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1062","ingestRunId":"launch-nvidia","modelId":"qwen3.5-397b-a17b","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"IMOAnswerBench (with tools)","locator":"Performance table: IMOAnswerBench (with tools)","metric":"%","notes":"First-party reported result. Original cell: 84.51","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":84.51,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"apex-shortlist-no-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"apex-shortlist-no-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1063","ingestRunId":"launch-nvidia","modelId":"nemotron-3-ultra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"Apex-Shortlist (no tools)","locator":"Performance table: Apex-Shortlist (no tools)","metric":"%","notes":"First-party reported result. Original cell: 74.9","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":74.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"apex-shortlist-no-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"apex-shortlist-no-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1064","ingestRunId":"launch-nvidia","modelId":"minimax-m2.7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"Apex-Shortlist (no tools)","locator":"Performance table: Apex-Shortlist (no tools)","metric":"%","notes":"First-party reported result. Original cell: 28.9","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":28.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"apex-shortlist-no-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"apex-shortlist-no-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1065","ingestRunId":"launch-nvidia","modelId":"glm-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"Apex-Shortlist (no tools)","locator":"Performance table: Apex-Shortlist (no tools)","metric":"%","notes":"First-party reported result. Original cell: 71.1","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":71.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"apex-shortlist-no-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"apex-shortlist-no-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1066","ingestRunId":"launch-nvidia","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"Apex-Shortlist (no tools)","locator":"Performance table: Apex-Shortlist (no tools)","metric":"%","notes":"First-party reported result. Original cell: 77.4","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":77.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"apex-shortlist-no-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"apex-shortlist-no-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1067","ingestRunId":"launch-nvidia","modelId":"qwen3.5-397b-a17b","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"Apex-Shortlist (no tools)","locator":"Performance table: Apex-Shortlist (no tools)","metric":"%","notes":"First-party reported result. Original cell: 61.4","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":61.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"apex-shortlist-with-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"apex-shortlist-with-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1068","ingestRunId":"launch-nvidia","modelId":"nemotron-3-ultra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"Apex-Shortlist (with tools)","locator":"Performance table: Apex-Shortlist (with tools)","metric":"%","notes":"First-party reported result. Original cell: 84.8","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":84.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"apex-shortlist-with-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"apex-shortlist-with-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1069","ingestRunId":"launch-nvidia","modelId":"minimax-m2.7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"Apex-Shortlist (with tools)","locator":"Performance table: Apex-Shortlist (with tools)","metric":"%","notes":"First-party reported result. Original cell: 51.9","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":51.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"apex-shortlist-with-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"apex-shortlist-with-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1070","ingestRunId":"launch-nvidia","modelId":"glm-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"Apex-Shortlist (with tools)","locator":"Performance table: Apex-Shortlist (with tools)","metric":"%","notes":"First-party reported result. Original cell: 79.0","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":79,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"apex-shortlist-with-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"apex-shortlist-with-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1071","ingestRunId":"launch-nvidia","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"Apex-Shortlist (with tools)","locator":"Performance table: Apex-Shortlist (with tools)","metric":"%","notes":"First-party reported result. Original cell: 73.2","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":73.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"apex-shortlist-with-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"apex-shortlist-with-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1072","ingestRunId":"launch-nvidia","modelId":"qwen3.5-397b-a17b","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"Apex-Shortlist (with tools)","locator":"Performance table: Apex-Shortlist (with tools)","metric":"%","notes":"First-party reported result. Original cell: 60.4","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":60.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"gpqa-no-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gpqa-no-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1073","ingestRunId":"launch-nvidia","modelId":"nemotron-3-ultra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"GPQA (no tools)","locator":"Performance table: GPQA (no tools)","metric":"%","notes":"First-party reported result. Original cell: 87.0 The linked NVIDIA reproduction recipe explicitly enables thinking for this model and benchmark. It does not establish Low/Medium/High/Max or a fixed reasoning-token budget; the request adapter removes max_tokens and max_completion_tokens. Comparator settings are not specified by this recipe.","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":87,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"gpqa-no-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gpqa-no-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1074","ingestRunId":"launch-nvidia","modelId":"minimax-m2.7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"GPQA (no tools)","locator":"Performance table: GPQA (no tools)","metric":"%","notes":"First-party reported result. Original cell: 86.6","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":86.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"gpqa-no-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gpqa-no-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1075","ingestRunId":"launch-nvidia","modelId":"glm-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"GPQA (no tools)","locator":"Performance table: GPQA (no tools)","metric":"%","notes":"First-party reported result. Original cell: 86.1","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":86.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"gpqa-no-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gpqa-no-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1076","ingestRunId":"launch-nvidia","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"GPQA (no tools)","locator":"Performance table: GPQA (no tools)","metric":"%","notes":"First-party reported result. Original cell: 91.0","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":91,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"gpqa-no-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gpqa-no-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1077","ingestRunId":"launch-nvidia","modelId":"qwen3.5-397b-a17b","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"GPQA (no tools)","locator":"Performance table: GPQA (no tools)","metric":"%","notes":"First-party reported result. Original cell: 87.1","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":87.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"scicode-subtask-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"scicode-subtask-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1078","ingestRunId":"launch-nvidia","modelId":"nemotron-3-ultra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"SciCode (subtask)","locator":"Performance table: SciCode (subtask)","metric":"%","notes":"First-party reported result. Original cell: 44.6 The linked NVIDIA reproduction recipe explicitly enables thinking for this model and benchmark. It does not establish Low/Medium/High/Max or a fixed reasoning-token budget; the request adapter removes max_tokens and max_completion_tokens. Comparator settings are not specified by this recipe.","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":44.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"scicode-subtask-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"scicode-subtask-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1079","ingestRunId":"launch-nvidia","modelId":"minimax-m2.7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"SciCode (subtask)","locator":"Performance table: SciCode (subtask)","metric":"%","notes":"First-party reported result. Original cell: 38.3","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":38.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"scicode-subtask-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"scicode-subtask-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1080","ingestRunId":"launch-nvidia","modelId":"glm-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"SciCode (subtask)","locator":"Performance table: SciCode (subtask)","metric":"%","notes":"First-party reported result. Original cell: 47.7","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":47.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"scicode-subtask-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"scicode-subtask-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1081","ingestRunId":"launch-nvidia","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"SciCode (subtask)","locator":"Performance table: SciCode (subtask)","metric":"%","notes":"First-party reported result. Original cell: 52.0","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":52,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"scicode-subtask-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"scicode-subtask-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1082","ingestRunId":"launch-nvidia","modelId":"qwen3.5-397b-a17b","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"SciCode (subtask)","locator":"Performance table: SciCode (subtask)","metric":"%","notes":"First-party reported result. Original cell: 48.0","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":48,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"hle-no-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hle-no-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1083","ingestRunId":"launch-nvidia","modelId":"nemotron-3-ultra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"Humanity's Last Exam","locator":"Performance table: HLE (no tools)","metric":"%","notes":"First-party reported result. Original cell: 26.7","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":26.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"hle-no-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hle-no-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1084","ingestRunId":"launch-nvidia","modelId":"minimax-m2.7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"Humanity's Last Exam","locator":"Performance table: HLE (no tools)","metric":"%","notes":"First-party reported result. Original cell: 23.1","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":23.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"hle-no-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hle-no-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1085","ingestRunId":"launch-nvidia","modelId":"glm-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"Humanity's Last Exam","locator":"Performance table: HLE (no tools)","metric":"%","notes":"First-party reported result. Original cell: 27.2","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":27.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"hle-no-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hle-no-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1086","ingestRunId":"launch-nvidia","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"Humanity's Last Exam","locator":"Performance table: HLE (no tools)","metric":"%","notes":"First-party reported result. Original cell: 34.8","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":34.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"hle-no-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hle-no-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1087","ingestRunId":"launch-nvidia","modelId":"qwen3.5-397b-a17b","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"Humanity's Last Exam","locator":"Performance table: HLE (no tools)","metric":"%","notes":"First-party reported result. Original cell: 28.5","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":28.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"hle-with-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hle-with-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1088","ingestRunId":"launch-nvidia","modelId":"nemotron-3-ultra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"Humanity's Last Exam","locator":"Performance table: HLE (with tools)","metric":"%","notes":"First-party reported result. Original cell: 37.4","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":37.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"hle-with-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hle-with-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1089","ingestRunId":"launch-nvidia","modelId":"glm-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"Humanity's Last Exam","locator":"Performance table: HLE (with tools)","metric":"%","notes":"First-party reported result. Original cell: 50.4","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":50.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"hle-with-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hle-with-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1090","ingestRunId":"launch-nvidia","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"Humanity's Last Exam","locator":"Performance table: HLE (with tools)","metric":"%","notes":"First-party reported result. Original cell: 54.0","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":54,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"hle-with-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hle-with-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1091","ingestRunId":"launch-nvidia","modelId":"qwen3.5-397b-a17b","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"Humanity's Last Exam","locator":"Performance table: HLE (with tools)","metric":"%","notes":"First-party reported result. Original cell: 48.3","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":48.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"critpt-no-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"critpt-no-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1092","ingestRunId":"launch-nvidia","modelId":"nemotron-3-ultra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"CritPt (no tools)","locator":"Performance table: CritPt (no tools)","metric":"%","notes":"First-party reported result. Original cell: 3.1 The linked NVIDIA reproduction recipe explicitly enables thinking for this model and benchmark. It does not establish Low/Medium/High/Max or a fixed reasoning-token budget; the request adapter removes max_tokens and max_completion_tokens. Comparator settings are not specified by this recipe.","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":3.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"critpt-no-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"critpt-no-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1093","ingestRunId":"launch-nvidia","modelId":"minimax-m2.7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"CritPt (no tools)","locator":"Performance table: CritPt (no tools)","metric":"%","notes":"First-party reported result. Original cell: 0.6","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":0.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"critpt-no-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"critpt-no-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1094","ingestRunId":"launch-nvidia","modelId":"glm-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"CritPt (no tools)","locator":"Performance table: CritPt (no tools)","metric":"%","notes":"First-party reported result. Original cell: 3.7","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":3.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"critpt-no-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"critpt-no-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1095","ingestRunId":"launch-nvidia","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"CritPt (no tools)","locator":"Performance table: CritPt (no tools)","metric":"%","notes":"First-party reported result. Original cell: 9.1","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":9.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"critpt-no-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"critpt-no-tools-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1096","ingestRunId":"launch-nvidia","modelId":"qwen3.5-397b-a17b","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"CritPt (no tools)","locator":"Performance table: CritPt (no tools)","metric":"%","notes":"First-party reported result. Original cell: 2.4","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":2.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"mmlu-pro-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"mmlu-pro-source-release-snapshot-version-not-specified-pct:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1097","ingestRunId":"launch-nvidia","modelId":"nemotron-3-ultra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"MMLU-Pro","locator":"Performance table: MMLU-Pro","metric":"%","notes":"First-party reported result. Original cell: 86.8 The linked NVIDIA reproduction recipe explicitly enables thinking for this model and benchmark. It does not establish Low/Medium/High/Max or a fixed reasoning-token budget; the request adapter removes max_tokens and max_completion_tokens. Comparator settings are not specified by this recipe.","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":86.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"mmlu-pro-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"mmlu-pro-source-release-snapshot-version-not-specified-pct:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1098","ingestRunId":"launch-nvidia","modelId":"minimax-m2.7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"MMLU-Pro","locator":"Performance table: MMLU-Pro","metric":"%","notes":"First-party reported result. Original cell: 81.9","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":81.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"mmlu-pro-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"mmlu-pro-source-release-snapshot-version-not-specified-pct:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1099","ingestRunId":"launch-nvidia","modelId":"glm-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"MMLU-Pro","locator":"Performance table: MMLU-Pro","metric":"%","notes":"First-party reported result. Original cell: 85.9","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":85.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"mmlu-pro-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"mmlu-pro-source-release-snapshot-version-not-specified-pct:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1100","ingestRunId":"launch-nvidia","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"MMLU-Pro","locator":"Performance table: MMLU-Pro","metric":"%","notes":"First-party reported result. Original cell: 88.1","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":88.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"mmlu-pro-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"mmlu-pro-source-release-snapshot-version-not-specified-pct:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1101","ingestRunId":"launch-nvidia","modelId":"qwen3.5-397b-a17b","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"MMLU-Pro","locator":"Performance table: MMLU-Pro","metric":"%","notes":"First-party reported result. Original cell: 88.3","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":88.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"omniscience-accuracy-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"omniscience-accuracy-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1102","ingestRunId":"launch-nvidia","modelId":"nemotron-3-ultra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"OmniScience Accuracy","locator":"Performance table: OmniScience Accuracy","metric":"%","notes":"First-party reported result. Original cell: 24.1 The linked NVIDIA reproduction recipe explicitly enables thinking for this model and benchmark. It does not establish Low/Medium/High/Max or a fixed reasoning-token budget; the request adapter removes max_tokens and max_completion_tokens. Comparator settings are not specified by this recipe.","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":24.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"omniscience-accuracy-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"omniscience-accuracy-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1103","ingestRunId":"launch-nvidia","modelId":"minimax-m2.7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"OmniScience Accuracy","locator":"Performance table: OmniScience Accuracy","metric":"%","notes":"First-party reported result. Original cell: 20.5","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":20.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"omniscience-accuracy-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"omniscience-accuracy-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1104","ingestRunId":"launch-nvidia","modelId":"glm-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"OmniScience Accuracy","locator":"Performance table: OmniScience Accuracy","metric":"%","notes":"First-party reported result. Original cell: 31.3","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":31.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"omniscience-accuracy-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"omniscience-accuracy-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1105","ingestRunId":"launch-nvidia","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"OmniScience Accuracy","locator":"Performance table: OmniScience Accuracy","metric":"%","notes":"First-party reported result. Original cell: 35.5","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":35.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"omniscience-accuracy-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"omniscience-accuracy-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1106","ingestRunId":"launch-nvidia","modelId":"qwen3.5-397b-a17b","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"OmniScience Accuracy","locator":"Performance table: OmniScience Accuracy","metric":"%","notes":"First-party reported result. Original cell: 35.9","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":35.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"ifbench-prompt-loose-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"ifbench-prompt-loose-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1107","ingestRunId":"launch-nvidia","modelId":"nemotron-3-ultra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"IFBench (prompt loose)","locator":"Performance table: IFBench (prompt loose)","metric":"%","notes":"First-party reported result. Original cell: 81.7 The linked NVIDIA reproduction recipe explicitly enables thinking for this model and benchmark. It does not establish Low/Medium/High/Max or a fixed reasoning-token budget; the request adapter removes max_tokens and max_completion_tokens. Comparator settings are not specified by this recipe.","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":81.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"ifbench-prompt-loose-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"ifbench-prompt-loose-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1108","ingestRunId":"launch-nvidia","modelId":"minimax-m2.7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"IFBench (prompt loose)","locator":"Performance table: IFBench (prompt loose)","metric":"%","notes":"First-party reported result. Original cell: 74.6","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":74.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"ifbench-prompt-loose-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"ifbench-prompt-loose-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1109","ingestRunId":"launch-nvidia","modelId":"glm-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"IFBench (prompt loose)","locator":"Performance table: IFBench (prompt loose)","metric":"%","notes":"First-party reported result. Original cell: 76.6","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":76.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"ifbench-prompt-loose-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"ifbench-prompt-loose-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1110","ingestRunId":"launch-nvidia","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"IFBench (prompt loose)","locator":"Performance table: IFBench (prompt loose)","metric":"%","notes":"First-party reported result. Original cell: 73.7","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":73.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"ifbench-prompt-loose-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"ifbench-prompt-loose-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1111","ingestRunId":"launch-nvidia","modelId":"qwen3.5-397b-a17b","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"IFBench (prompt loose)","locator":"Performance table: IFBench (prompt loose)","metric":"%","notes":"First-party reported result. Original cell: 78.2","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":78.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"multi-challenge-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"multi-challenge-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1112","ingestRunId":"launch-nvidia","modelId":"nemotron-3-ultra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"Multi-Challenge","locator":"Performance table: Multi-Challenge","metric":"%","notes":"First-party reported result. Original cell: 63.8 The linked NVIDIA reproduction recipe explicitly enables thinking for this model and benchmark. It does not establish Low/Medium/High/Max or a fixed reasoning-token budget; the request adapter removes max_tokens and max_completion_tokens. Comparator settings are not specified by this recipe.","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":63.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"multi-challenge-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"multi-challenge-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1113","ingestRunId":"launch-nvidia","modelId":"minimax-m2.7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"Multi-Challenge","locator":"Performance table: Multi-Challenge","metric":"%","notes":"First-party reported result. Original cell: 42.5","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":42.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"multi-challenge-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"multi-challenge-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1114","ingestRunId":"launch-nvidia","modelId":"glm-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"Multi-Challenge","locator":"Performance table: Multi-Challenge","metric":"%","notes":"First-party reported result. Original cell: 63.0","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":63,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"multi-challenge-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"multi-challenge-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1115","ingestRunId":"launch-nvidia","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"Multi-Challenge","locator":"Performance table: Multi-Challenge","metric":"%","notes":"First-party reported result. Original cell: 63.1","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":63.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"multi-challenge-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"multi-challenge-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1116","ingestRunId":"launch-nvidia","modelId":"qwen3.5-397b-a17b","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"human_pref","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"Multi-Challenge","locator":"Performance table: Multi-Challenge","metric":"%","notes":"First-party reported result. Original cell: 63.9","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":63.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"aa-lcr-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"aa-lcr-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1117","ingestRunId":"launch-nvidia","modelId":"nemotron-3-ultra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"AA-LCR","locator":"Performance table: AA-LCR","metric":"%","notes":"First-party reported result. Original cell: 65.4 The linked NVIDIA reproduction recipe explicitly enables thinking for this model and benchmark. It does not establish Low/Medium/High/Max or a fixed reasoning-token budget; the request adapter removes max_tokens and max_completion_tokens. Comparator settings are not specified by this recipe.","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":65.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"aa-lcr-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"aa-lcr-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1118","ingestRunId":"launch-nvidia","modelId":"minimax-m2.7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"AA-LCR","locator":"Performance table: AA-LCR","metric":"%","notes":"First-party reported result. Original cell: 69.8","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":69.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"aa-lcr-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"aa-lcr-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1119","ingestRunId":"launch-nvidia","modelId":"glm-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"AA-LCR","locator":"Performance table: AA-LCR","metric":"%","notes":"First-party reported result. Original cell: 66.9","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":66.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"aa-lcr-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"aa-lcr-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1120","ingestRunId":"launch-nvidia","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"AA-LCR","locator":"Performance table: AA-LCR","metric":"%","notes":"First-party reported result. Original cell: 70.2","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":70.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"aa-lcr-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"aa-lcr-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1121","ingestRunId":"launch-nvidia","modelId":"qwen3.5-397b-a17b","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"AA-LCR","locator":"Performance table: AA-LCR","metric":"%","notes":"First-party reported result. Original cell: 68.3","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":68.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"ruler-1m-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"ruler-1m-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1122","ingestRunId":"launch-nvidia","modelId":"nemotron-3-ultra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"RULER (1M)","locator":"Performance table: RULER (1M)","metric":"%","notes":"First-party reported result. Original cell: 94.7","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":94.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"ruler-1m-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"ruler-1m-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1123","ingestRunId":"launch-nvidia","modelId":"qwen3.5-397b-a17b","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"RULER (1M)","locator":"Performance table: RULER (1M)","metric":"%","notes":"First-party reported result. Original cell: 90.1","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":90.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"longbench-v2-lte-1m-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"longbench-v2-lte-1m-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1124","ingestRunId":"launch-nvidia","modelId":"nemotron-3-ultra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"Longbench v2 (≤ 1M)","locator":"Performance table: Longbench v2 (≤ 1M)","metric":"%","notes":"First-party reported result. Original cell: 61.9","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":61.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"longbench-v2-lte-1m-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"longbench-v2-lte-1m-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1125","ingestRunId":"launch-nvidia","modelId":"qwen3.5-397b-a17b","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"Longbench v2 (≤ 1M)","locator":"Performance table: Longbench v2 (≤ 1M)","metric":"%","notes":"First-party reported result. Original cell: 68.9","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":68.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"mmlu-prox-avg-en-de-fr-es-it-ja-zh-hi-pt-ko-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mmlu-prox-avg-en-de-fr-es-it-ja-zh-hi-pt-ko-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1126","ingestRunId":"launch-nvidia","modelId":"nemotron-3-ultra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"MMLU-ProX (avg en/de/fr/es/it/ja/zh/hi/pt/ko)","locator":"Performance table: MMLU-ProX (avg en/de/fr/es/it/ja/zh/hi/pt/ko)","metric":"%","notes":"First-party reported result. Original cell: 83.0 The linked NVIDIA reproduction recipe explicitly enables thinking for this model and benchmark. It does not establish Low/Medium/High/Max or a fixed reasoning-token budget; the request adapter removes max_tokens and max_completion_tokens. Comparator settings are not specified by this recipe.","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":83,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"mmlu-prox-avg-en-de-fr-es-it-ja-zh-hi-pt-ko-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mmlu-prox-avg-en-de-fr-es-it-ja-zh-hi-pt-ko-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1127","ingestRunId":"launch-nvidia","modelId":"minimax-m2.7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"MMLU-ProX (avg en/de/fr/es/it/ja/zh/hi/pt/ko)","locator":"Performance table: MMLU-ProX (avg en/de/fr/es/it/ja/zh/hi/pt/ko)","metric":"%","notes":"First-party reported result. Original cell: 78.4","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":78.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"mmlu-prox-avg-en-de-fr-es-it-ja-zh-hi-pt-ko-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mmlu-prox-avg-en-de-fr-es-it-ja-zh-hi-pt-ko-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1128","ingestRunId":"launch-nvidia","modelId":"glm-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"MMLU-ProX (avg en/de/fr/es/it/ja/zh/hi/pt/ko)","locator":"Performance table: MMLU-ProX (avg en/de/fr/es/it/ja/zh/hi/pt/ko)","metric":"%","notes":"First-party reported result. Original cell: 85.8","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":85.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"mmlu-prox-avg-en-de-fr-es-it-ja-zh-hi-pt-ko-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mmlu-prox-avg-en-de-fr-es-it-ja-zh-hi-pt-ko-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1129","ingestRunId":"launch-nvidia","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"MMLU-ProX (avg en/de/fr/es/it/ja/zh/hi/pt/ko)","locator":"Performance table: MMLU-ProX (avg en/de/fr/es/it/ja/zh/hi/pt/ko)","metric":"%","notes":"First-party reported result. Original cell: 85.0","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":85,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"mmlu-prox-avg-en-de-fr-es-it-ja-zh-hi-pt-ko-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mmlu-prox-avg-en-de-fr-es-it-ja-zh-hi-pt-ko-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1130","ingestRunId":"launch-nvidia","modelId":"qwen3.5-397b-a17b","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"MMLU-ProX (avg en/de/fr/es/it/ja/zh/hi/pt/ko)","locator":"Performance table: MMLU-ProX (avg en/de/fr/es/it/ja/zh/hi/pt/ko)","metric":"%","notes":"First-party reported result. Original cell: 86.4","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":86.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"wmt24-en-to-xx-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"wmt24-en-to-xx-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1131","ingestRunId":"launch-nvidia","modelId":"nemotron-3-ultra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"WMT24++ (en→xx)","locator":"Performance table: WMT24++ (en→xx)","metric":"score (source scale)","notes":"First-party reported result. Original cell: 83.7 The linked NVIDIA reproduction recipe explicitly enables thinking for this model and benchmark. It does not establish Low/Medium/High/Max or a fixed reasoning-token budget; the request adapter removes max_tokens and max_completion_tokens. Comparator settings are not specified by this recipe.","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":83.7,"scoreUnit":"index","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"wmt24-en-to-xx-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"wmt24-en-to-xx-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1132","ingestRunId":"launch-nvidia","modelId":"minimax-m2.7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"WMT24++ (en→xx)","locator":"Performance table: WMT24++ (en→xx)","metric":"score (source scale)","notes":"First-party reported result. Original cell: 82.8","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":82.8,"scoreUnit":"index","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"wmt24-en-to-xx-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"wmt24-en-to-xx-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1133","ingestRunId":"launch-nvidia","modelId":"glm-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"WMT24++ (en→xx)","locator":"Performance table: WMT24++ (en→xx)","metric":"score (source scale)","notes":"First-party reported result. Original cell: 84.4","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":84.4,"scoreUnit":"index","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"wmt24-en-to-xx-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"wmt24-en-to-xx-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1134","ingestRunId":"launch-nvidia","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"WMT24++ (en→xx)","locator":"Performance table: WMT24++ (en→xx)","metric":"score (source scale)","notes":"First-party reported result. Original cell: 84.5","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":84.5,"scoreUnit":"index","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"wmt24-en-to-xx-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"wmt24-en-to-xx-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-1135","ingestRunId":"launch-nvidia","modelId":"qwen3.5-397b-a17b","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"WMT24++ (en→xx)","locator":"Performance table: WMT24++ (en→xx)","metric":"score (source scale)","notes":"First-party reported result. Original cell: 86.8","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":86.8,"scoreUnit":"index","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"terminal-bench-2-1","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-974","ingestRunId":"launch-nvidia","modelId":"nemotron-3-ultra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"Terminal-Bench","locator":"Performance table: Terminal Bench 2.1","metric":"%","notes":"First-party reported result. Original cell: 56.4 The linked public reproduction recipe names Terminal Bench 2.0, while this card labels 2.1. Recipe settings are not transferred across that unresolved version mismatch.","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":56.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"terminal-bench-2-1","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-975","ingestRunId":"launch-nvidia","modelId":"minimax-m2.7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"Terminal-Bench","locator":"Performance table: Terminal Bench 2.1","metric":"%","notes":"First-party reported result. Original cell: 55.5 The linked public reproduction recipe names Terminal Bench 2.0, while this card labels 2.1. Recipe settings are not transferred across that unresolved version mismatch.","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":55.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"terminal-bench-2-1","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-976","ingestRunId":"launch-nvidia","modelId":"glm-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"Terminal-Bench","locator":"Performance table: Terminal Bench 2.1","metric":"%","notes":"First-party reported result. Original cell: 59.3 The linked public reproduction recipe names Terminal Bench 2.0, while this card labels 2.1. Recipe settings are not transferred across that unresolved version mismatch.","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":59.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"terminal-bench-2-1","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-977","ingestRunId":"launch-nvidia","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"Terminal-Bench","locator":"Performance table: Terminal Bench 2.1","metric":"%","notes":"First-party reported result. Original cell: 67.2 The linked public reproduction recipe names Terminal Bench 2.0, while this card labels 2.1. Recipe settings are not transferred across that unresolved version mismatch.","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":67.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"terminal-bench-2-1","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-978","ingestRunId":"launch-nvidia","modelId":"qwen3.5-397b-a17b","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"Terminal-Bench","locator":"Performance table: Terminal Bench 2.1","metric":"%","notes":"First-party reported result. Original cell: 49.9 The linked public reproduction recipe names Terminal Bench 2.0, while this card labels 2.1. Recipe settings are not transferred across that unresolved version mismatch.","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":49.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"gdpval-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gdpval-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-979","ingestRunId":"launch-nvidia","modelId":"nemotron-3-ultra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"GDPVal","locator":"Performance table: GDPVal","metric":"%","notes":"First-party reported result. Original cell: 46.7 The linked NVIDIA reproduction recipe explicitly enables thinking for this model and benchmark. It does not establish Low/Medium/High/Max or a fixed reasoning-token budget; the request adapter removes max_tokens and max_completion_tokens. Comparator settings are not specified by this recipe.","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":46.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"gdpval-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gdpval-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-980","ingestRunId":"launch-nvidia","modelId":"minimax-m2.7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"GDPVal","locator":"Performance table: GDPVal","metric":"%","notes":"First-party reported result. Original cell: 47.6","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":47.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"gdpval-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gdpval-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-981","ingestRunId":"launch-nvidia","modelId":"glm-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"GDPVal","locator":"Performance table: GDPVal","metric":"%","notes":"First-party reported result. Original cell: 54.7","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":54.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"gdpval-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gdpval-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-982","ingestRunId":"launch-nvidia","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"GDPVal","locator":"Performance table: GDPVal","metric":"%","notes":"First-party reported result. Original cell: 50.4","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":50.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"gdpval-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gdpval-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-983","ingestRunId":"launch-nvidia","modelId":"qwen3.5-397b-a17b","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"GDPVal","locator":"Performance table: GDPVal","metric":"%","notes":"First-party reported result. Original cell: 34.6","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":34.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"swe-bench-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"swe-bench-verified-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-984","ingestRunId":"launch-nvidia","modelId":"nemotron-3-ultra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"SWE-bench Verified","locator":"Performance table: SWE-Bench Verified","metric":"%","notes":"First-party reported result. Original cell: 70.7 The linked NVIDIA reproduction recipe explicitly enables thinking for this model and benchmark. It does not establish Low/Medium/High/Max or a fixed reasoning-token budget; the request adapter removes max_tokens and max_completion_tokens. Comparator settings are not specified by this recipe.","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":70.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"swe-bench-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"swe-bench-verified-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-985","ingestRunId":"launch-nvidia","modelId":"minimax-m2.7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"SWE-bench Verified","locator":"Performance table: SWE-Bench Verified","metric":"%","notes":"First-party reported result. Original cell: 75.3","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":75.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"swe-bench-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"swe-bench-verified-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-986","ingestRunId":"launch-nvidia","modelId":"glm-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"SWE-bench Verified","locator":"Performance table: SWE-Bench Verified","metric":"%","notes":"First-party reported result. Original cell: 76.2","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":76.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"swe-bench-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"swe-bench-verified-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-987","ingestRunId":"launch-nvidia","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"SWE-bench Verified","locator":"Performance table: SWE-Bench Verified","metric":"%","notes":"First-party reported result. Original cell: 75.7","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":75.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"swe-bench-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"swe-bench-verified-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-988","ingestRunId":"launch-nvidia","modelId":"qwen3.5-397b-a17b","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"SWE-bench Verified","locator":"Performance table: SWE-Bench Verified","metric":"%","notes":"First-party reported result. Original cell: 73.6","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":73.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"swe-bench-multilingual-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"swe-bench-multilingual-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-989","ingestRunId":"launch-nvidia","modelId":"nemotron-3-ultra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"SWE-bench Multilingual","locator":"Performance table: SWE-Bench Multilingual","metric":"%","notes":"First-party reported result. Original cell: 67.7","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":67.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"swe-bench-multilingual-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"swe-bench-multilingual-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-990","ingestRunId":"launch-nvidia","modelId":"minimax-m2.7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"SWE-bench Multilingual","locator":"Performance table: SWE-Bench Multilingual","metric":"%","notes":"First-party reported result. Original cell: 71.8","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":71.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"swe-bench-multilingual-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"swe-bench-multilingual-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-991","ingestRunId":"launch-nvidia","modelId":"glm-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"SWE-bench Multilingual","locator":"Performance table: SWE-Bench Multilingual","metric":"%","notes":"First-party reported result. Original cell: 74.8","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":74.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"swe-bench-multilingual-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"swe-bench-multilingual-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-992","ingestRunId":"launch-nvidia","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"SWE-bench Multilingual","locator":"Performance table: SWE-Bench Multilingual","metric":"%","notes":"First-party reported result. Original cell: 77.1","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":77.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"swe-bench-multilingual-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"swe-bench-multilingual-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-993","ingestRunId":"launch-nvidia","modelId":"qwen3.5-397b-a17b","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"SWE-bench Multilingual","locator":"Performance table: SWE-Bench Multilingual","metric":"%","notes":"First-party reported result. Original cell: 70.9","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":70.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"profbench-search-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"profbench-search-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-994","ingestRunId":"launch-nvidia","modelId":"nemotron-3-ultra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"ProfBench (Search)","locator":"Performance table: ProfBench (Search)","metric":"%","notes":"First-party reported result. Original cell: 56.0","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":56,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"profbench-search-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"profbench-search-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-995","ingestRunId":"launch-nvidia","modelId":"minimax-m2.7","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"ProfBench (Search)","locator":"Performance table: ProfBench (Search)","metric":"%","notes":"First-party reported result. Original cell: 52.0","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":52,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"profbench-search-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"profbench-search-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-996","ingestRunId":"launch-nvidia","modelId":"glm-5.1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"ProfBench (Search)","locator":"Performance table: ProfBench (Search)","metric":"%","notes":"First-party reported result. Original cell: 46.0","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":46,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"profbench-search-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"profbench-search-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-997","ingestRunId":"launch-nvidia","modelId":"kimi-k2.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"ProfBench (Search)","locator":"Performance table: ProfBench (Search)","metric":"%","notes":"First-party reported result. Original cell: 56.0","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":56,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"profbench-search-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"profbench-search-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-998","ingestRunId":"launch-nvidia","modelId":"qwen3.5-397b-a17b","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"ProfBench (Search)","locator":"Performance table: ProfBench (Search)","metric":"%","notes":"First-party reported result. Original cell: 53.0","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":53,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"pinchbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"pinchbench-source-release-snapshot-version-not-specified:nvidia:TlZJRElBIGV2YWx1YXRpb24gaGFybmVzcy9zZXR0aW5ncyBwZXIgYmVuY2htYXJrIGluIG1vZGVsIGNhcmQ7IGNvbXBhcmF0b3Igc2NvcmVzIGFyZSBOVklESUEtcmVwb3J0ZWQgdW5kZXIgdGhhdCBwcm90b2NvbC4","id":"launch-nvidia-999","ingestRunId":"launch-nvidia","modelId":"nemotron-3-ultra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol.","family":"PinchBench","locator":"Performance table: PinchBench","metric":"%","notes":"First-party reported result. Original cell: 90.0","sourceId":"nvidia","sourceTitle":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},"score":90,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"},{"benchmarkId":"agents-last-exam-not-specified","evidenceKind":"lab_self_report","harnessId":"agents-last-exam-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-137","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"Agents' Last Exam","locator":"Computer Use table / Agents' Last Exam / GPT‑6 Astra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":59.3,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"agents-last-exam-not-specified","evidenceKind":"lab_self_report","harnessId":"agents-last-exam-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-138","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"Agents' Last Exam","locator":"Computer Use table / Agents' Last Exam / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":53.6,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"agents-last-exam-not-specified","evidenceKind":"lab_self_report","harnessId":"agents-last-exam-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-139","ingestRunId":"launch-openai-astra-launch","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"Agents' Last Exam","locator":"Computer Use table / Agents' Last Exam / Claude Fable 5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":48.7,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"agents-last-exam-not-specified","evidenceKind":"lab_self_report","harnessId":"agents-last-exam-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-140","ingestRunId":"launch-openai-astra-launch","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"Agents' Last Exam","locator":"Computer Use table / Agents' Last Exam / Claude Opus 5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":55.5,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline","evidenceKind":"lab_self_report","harnessId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IG9mZmxpbmUgc2V0OyBwYXJ0aWFsIGNyZWRpdDsgdjIwMjYuMDguMDg7IG9mZmljaWFsIHRhc2svZ3JhZGluZyBzZXR0aW5ncw","id":"launch-openai-astra-launch-141","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; offline set; partial credit; v2026.08.08; official task/grading settings","family":"OSWorld","locator":"Computer Use table / OSWorld 2.0 (v2026.08.08, offline set, partial score) / GPT‑6 Astra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":72.6,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline","evidenceKind":"lab_self_report","harnessId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IG9mZmxpbmUgc2V0OyBwYXJ0aWFsIGNyZWRpdDsgdjIwMjYuMDguMDg7IG9mZmljaWFsIHRhc2svZ3JhZGluZyBzZXR0aW5ncw","id":"launch-openai-astra-launch-142","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; offline set; partial credit; v2026.08.08; official task/grading settings","family":"OSWorld","locator":"Computer Use table / OSWorld 2.0 (v2026.08.08, offline set, partial score) / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":65.7,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline","evidenceKind":"lab_self_report","harnessId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IG9mZmxpbmUgc2V0OyBwYXJ0aWFsIGNyZWRpdDsgdjIwMjYuMDguMDg7IG9mZmljaWFsIHRhc2svZ3JhZGluZyBzZXR0aW5ncw","id":"launch-openai-astra-launch-143","ingestRunId":"launch-openai-astra-launch","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; offline set; partial credit; v2026.08.08; official task/grading settings","family":"OSWorld","locator":"Computer Use table / OSWorld 2.0 (v2026.08.08, offline set, partial score) / Claude Opus 5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":70.2,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"screenspot-pro-no-tools-not-specified","evidenceKind":"lab_self_report","harnessId":"screenspot-pro-no-tools-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IG5vIHRvb2xz","id":"launch-openai-astra-launch-144","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; no tools","family":"ScreenSpot-Pro","locator":"Computer Use table / ScreenSpot-Pro (no tools) / GPT‑6 Astra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":92.7,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"screenspot-pro-no-tools-not-specified","evidenceKind":"lab_self_report","harnessId":"screenspot-pro-no-tools-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IG5vIHRvb2xz","id":"launch-openai-astra-launch-145","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; no tools","family":"ScreenSpot-Pro","locator":"Computer Use table / ScreenSpot-Pro (no tools) / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":76.9,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"automationbench-not-specified","evidenceKind":"lab_self_report","harnessId":"automationbench-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-146","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"AutomationBench","locator":"Professional table / AutomationBench / GPT‑6 Astra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":41.4,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"automationbench-not-specified","evidenceKind":"lab_self_report","harnessId":"automationbench-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-147","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"AutomationBench","locator":"Professional table / AutomationBench / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":18.1,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"automationbench-not-specified","evidenceKind":"lab_self_report","harnessId":"automationbench-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-148","ingestRunId":"launch-openai-astra-launch","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"AutomationBench","locator":"Professional table / AutomationBench / Claude Fable 5.1","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":31.4,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"automationbench-not-specified","evidenceKind":"lab_self_report","harnessId":"automationbench-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-149","ingestRunId":"launch-openai-astra-launch","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"AutomationBench","locator":"Professional table / AutomationBench / Claude Fable 5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":17.4,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"automationbench-not-specified","evidenceKind":"lab_self_report","harnessId":"automationbench-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-150","ingestRunId":"launch-openai-astra-launch","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"AutomationBench","locator":"Professional table / AutomationBench / Claude Opus 5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":26.9,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"benchcad-not-specified","evidenceKind":"lab_self_report","harnessId":"benchcad-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHRvb2xzIGVuYWJsZWQ","id":"launch-openai-astra-launch-151","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; tools enabled","family":"BenchCAD","locator":"Professional table / BenchCAD / GPT‑6 Astra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":95.9,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"benchcad-not-specified","evidenceKind":"lab_self_report","harnessId":"benchcad-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHRvb2xzIGVuYWJsZWQ","id":"launch-openai-astra-launch-152","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; tools enabled","family":"BenchCAD","locator":"Professional table / BenchCAD / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":83.3,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"benchcad-not-specified","evidenceKind":"lab_self_report","harnessId":"benchcad-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHRvb2xzIGVuYWJsZWQ7IHRocmVlIEFudGhyb3BpYyBldmFsdWF0aW9uIG1vZGlmaWNhdGlvbnM","id":"launch-openai-astra-launch-153","ingestRunId":"launch-openai-astra-launch","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; tools enabled; three Anthropic evaluation modifications","family":"BenchCAD","locator":"Professional table / BenchCAD / Claude Fable 5.1","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced. Astra launch footnote5: Claude scores use three modifications described in the Fable5.1 system card; not same controlled setting.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":84.3,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"benchcad-not-specified","evidenceKind":"lab_self_report","harnessId":"benchcad-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHRvb2xzIGVuYWJsZWQ7IHRocmVlIEFudGhyb3BpYyBldmFsdWF0aW9uIG1vZGlmaWNhdGlvbnM","id":"launch-openai-astra-launch-154","ingestRunId":"launch-openai-astra-launch","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; tools enabled; three Anthropic evaluation modifications","family":"BenchCAD","locator":"Professional table / BenchCAD / Claude Fable 5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced. Astra launch footnote5: Claude scores use three modifications described in the Fable5.1 system card; not same controlled setting.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":67.5,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"benchcad-not-specified","evidenceKind":"lab_self_report","harnessId":"benchcad-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHRvb2xzIGVuYWJsZWQ7IHRocmVlIEFudGhyb3BpYyBldmFsdWF0aW9uIG1vZGlmaWNhdGlvbnM","id":"launch-openai-astra-launch-155","ingestRunId":"launch-openai-astra-launch","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; tools enabled; three Anthropic evaluation modifications","family":"BenchCAD","locator":"Professional table / BenchCAD / Claude Opus 5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced. Astra launch footnote5: Claude scores use three modifications described in the Fable5.1 system card; not same controlled setting.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":82.1,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"browsecomp-not-specified","evidenceKind":"lab_self_report","harnessId":"browsecomp-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-156","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"BrowseComp","locator":"Professional table / BrowseComp / GPT‑6 Astra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":91.5,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"browsecomp-not-specified","evidenceKind":"lab_self_report","harnessId":"browsecomp-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-157","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"BrowseComp","locator":"Professional table / BrowseComp / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":90.4,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"browsecomp-not-specified","evidenceKind":"lab_self_report","harnessId":"browsecomp-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-158","ingestRunId":"launch-openai-astra-launch","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"BrowseComp","locator":"Professional table / BrowseComp / Claude Fable 5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":87.4,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"browsecomp-not-specified","evidenceKind":"lab_self_report","harnessId":"browsecomp-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-159","ingestRunId":"launch-openai-astra-launch","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"BrowseComp","locator":"Professional table / BrowseComp / Claude Opus 5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":90.8,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"openscore-string-quartets-1-omr-ned-not-specified","evidenceKind":"lab_self_report","harnessId":"openscore-string-quartets-1-omr-ned-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-160","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"OpenScore String Quartets (1 - OMR-NED)","locator":"Professional table / OpenScore String Quartets (1 - OMR-NED) / GPT‑6 Astra","metric":"1 - OMR-NED","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":0.84,"scoreUnit":"index","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"openscore-string-quartets-1-omr-ned-not-specified","evidenceKind":"lab_self_report","harnessId":"openscore-string-quartets-1-omr-ned-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-161","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"OpenScore String Quartets (1 - OMR-NED)","locator":"Professional table / OpenScore String Quartets (1 - OMR-NED) / GPT‑5.6 Sol","metric":"1 - OMR-NED","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":0.19,"scoreUnit":"index","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"internal-design-tasks-not-specified","evidenceKind":"lab_self_report","harnessId":"internal-design-tasks-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-162","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"Internal Design Tasks","locator":"Professional table / Internal Design Tasks / GPT‑6 Astra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":50,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"internal-design-tasks-not-specified","evidenceKind":"lab_self_report","harnessId":"internal-design-tasks-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-163","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"Internal Design Tasks","locator":"Professional table / Internal Design Tasks / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":47.4,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"internal-design-tasks-not-specified","evidenceKind":"lab_self_report","harnessId":"internal-design-tasks-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-164","ingestRunId":"launch-openai-astra-launch","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"Internal Design Tasks","locator":"Professional table / Internal Design Tasks / Claude Fable 5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":35.8,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"internal-data-science-tasks-not-specified","evidenceKind":"lab_self_report","harnessId":"internal-data-science-tasks-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-165","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"Internal Data Science Tasks","locator":"Professional table / Internal Data Science Tasks / GPT‑6 Astra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":40.9,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"internal-data-science-tasks-not-specified","evidenceKind":"lab_self_report","harnessId":"internal-data-science-tasks-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-166","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"Internal Data Science Tasks","locator":"Professional table / Internal Data Science Tasks / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":30.5,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"internal-data-science-tasks-not-specified","evidenceKind":"lab_self_report","harnessId":"internal-data-science-tasks-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-167","ingestRunId":"launch-openai-astra-launch","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"Internal Data Science Tasks","locator":"Professional table / Internal Data Science Tasks / Claude Fable 5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":34.7,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"artificial-analysis-intelligence-index-v4-1-1-4-1-1","evidenceKind":"lab_self_report","harnessId":"artificial-analysis-intelligence-index-v4-1-1-4-1-1:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-168","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"Artificial Analysis Intelligence Index v4.1.1","locator":"Professional table / Artificial Analysis Intelligence Index v4.1.1 / GPT‑6 Astra","metric":"index score","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":61.2,"scoreUnit":"index","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"artificial-analysis-intelligence-index-v4-1-1-4-1-1","evidenceKind":"lab_self_report","harnessId":"artificial-analysis-intelligence-index-v4-1-1-4-1-1:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-169","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"Artificial Analysis Intelligence Index v4.1.1","locator":"Professional table / Artificial Analysis Intelligence Index v4.1.1 / GPT‑5.6 Sol","metric":"index score","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":60.9,"scoreUnit":"index","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"artificial-analysis-intelligence-index-v4-1-1-4-1-1","evidenceKind":"lab_self_report","harnessId":"artificial-analysis-intelligence-index-v4-1-1-4-1-1:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-170","ingestRunId":"launch-openai-astra-launch","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"Artificial Analysis Intelligence Index v4.1.1","locator":"Professional table / Artificial Analysis Intelligence Index v4.1.1 / Claude Fable 5.1","metric":"index score","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":65.7,"scoreUnit":"index","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"artificial-analysis-intelligence-index-v4-1-1-4-1-1","evidenceKind":"lab_self_report","harnessId":"artificial-analysis-intelligence-index-v4-1-1-4-1-1:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-171","ingestRunId":"launch-openai-astra-launch","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"Artificial Analysis Intelligence Index v4.1.1","locator":"Professional table / Artificial Analysis Intelligence Index v4.1.1 / Claude Fable 5","metric":"index score","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":62.1,"scoreUnit":"index","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"artificial-analysis-intelligence-index-v4-1-1-4-1-1","evidenceKind":"lab_self_report","harnessId":"artificial-analysis-intelligence-index-v4-1-1-4-1-1:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-172","ingestRunId":"launch-openai-astra-launch","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"Artificial Analysis Intelligence Index v4.1.1","locator":"Professional table / Artificial Analysis Intelligence Index v4.1.1 / Claude Opus 5","metric":"index score","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":63.1,"scoreUnit":"index","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"artificial-analysis-intelligence-index-v4-1-1-4-1-1","evidenceKind":"lab_self_report","harnessId":"artificial-analysis-intelligence-index-v4-1-1-4-1-1:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-173","ingestRunId":"launch-openai-astra-launch","modelId":"gemini-3.8-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"Artificial Analysis Intelligence Index v4.1.1","locator":"Professional table / Artificial Analysis Intelligence Index v4.1.1 / Gemini 3.8 Flash","metric":"index score","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":58.7,"scoreUnit":"index","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"terminal-bench-4-0","evidenceKind":"lab_self_report","harnessId":"terminal-bench-4-0:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-174","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"Terminal-Bench","locator":"Coding table / Terminal-Bench 4.0 / GPT‑6 Astra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":57.9,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"terminal-bench-4-0","evidenceKind":"lab_self_report","harnessId":"terminal-bench-4-0:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-175","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"Terminal-Bench","locator":"Coding table / Terminal-Bench 4.0 / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":37.3,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"terminal-bench-4-0","evidenceKind":"lab_self_report","harnessId":"terminal-bench-4-0:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-176","ingestRunId":"launch-openai-astra-launch","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"Terminal-Bench","locator":"Coding table / Terminal-Bench 4.0 / Claude Fable 5.1","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":55.8,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"terminal-bench-4-0","evidenceKind":"lab_self_report","harnessId":"terminal-bench-4-0:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-177","ingestRunId":"launch-openai-astra-launch","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"Terminal-Bench","locator":"Coding table / Terminal-Bench 4.0 / Claude Fable 5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":44.5,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"terminal-bench-4-0","evidenceKind":"lab_self_report","harnessId":"terminal-bench-4-0:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-178","ingestRunId":"launch-openai-astra-launch","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"Terminal-Bench","locator":"Coding table / Terminal-Bench 4.0 / Claude Opus 5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":52.6,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"terminal-bench-4-0","evidenceKind":"lab_self_report","harnessId":"terminal-bench-4-0:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-179","ingestRunId":"launch-openai-astra-launch","modelId":"gemini-3.8-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"Terminal-Bench","locator":"Coding table / Terminal-Bench 4.0 / Gemini 3.8 Flash","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":19.1,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"deepswe-v1-1-1-1-percent","evidenceKind":"lab_self_report","harnessId":"deepswe-v1-1-1-1-percent:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-180","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"DeepSWE v1.1","locator":"Coding table / DeepSWE v1.1 / GPT‑6 Astra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":74.1,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"deepswe-v1-1-1-1-percent","evidenceKind":"lab_self_report","harnessId":"deepswe-v1-1-1-1-percent:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-181","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"DeepSWE v1.1","locator":"Coding table / DeepSWE v1.1 / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":72.7,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"deepswe-v1-1-1-1-percent","evidenceKind":"lab_self_report","harnessId":"deepswe-v1-1-1-1-percent:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-182","ingestRunId":"launch-openai-astra-launch","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"DeepSWE v1.1","locator":"Coding table / DeepSWE v1.1 / Claude Fable 5.1","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":67.4,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"deepswe-v1-1-1-1-percent","evidenceKind":"lab_self_report","harnessId":"deepswe-v1-1-1-1-percent:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-183","ingestRunId":"launch-openai-astra-launch","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"DeepSWE v1.1","locator":"Coding table / DeepSWE v1.1 / Claude Fable 5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":69.9,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"deepswe-v1-1-1-1-percent","evidenceKind":"lab_self_report","harnessId":"deepswe-v1-1-1-1-percent:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-184","ingestRunId":"launch-openai-astra-launch","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"DeepSWE v1.1","locator":"Coding table / DeepSWE v1.1 / Claude Opus 5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":73.7,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"deepswe-v1-1-1-1-percent","evidenceKind":"lab_self_report","harnessId":"deepswe-v1-1-1-1-percent:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-185","ingestRunId":"launch-openai-astra-launch","modelId":"gemini-3.8-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"DeepSWE v1.1","locator":"Coding table / DeepSWE v1.1 / Gemini 3.8 Flash","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":73.8,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"frontiercode-1-1-extended-score-1-1","evidenceKind":"lab_self_report","harnessId":"frontiercode-1-1-extended-score-1-1:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IENvZGV4LXN0eWxlIGRldmVsb3BlciBpbnN0cnVjdGlvbiBvbiB0ZXN0cywgcmV1c2UgYW5kIHJlcG9zaXRvcnkgY29udmVudGlvbnM","id":"launch-openai-astra-launch-186","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; Codex-style developer instruction on tests, reuse and repository conventions","family":"FrontierCode 1.1 Extended (score)","locator":"Coding table / FrontierCode 1.1 Extended (score) / GPT‑6 Astra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":64.5,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"frontiercode-1-1-extended-score-1-1","evidenceKind":"lab_self_report","harnessId":"frontiercode-1-1-extended-score-1-1:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-187","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"FrontierCode 1.1 Extended (score)","locator":"Coding table / FrontierCode 1.1 Extended (score) / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":60.6,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"frontiercode-1-1-extended-score-1-1","evidenceKind":"lab_self_report","harnessId":"frontiercode-1-1-extended-score-1-1:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-188","ingestRunId":"launch-openai-astra-launch","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"FrontierCode 1.1 Extended (score)","locator":"Coding table / FrontierCode 1.1 Extended (score) / Claude Fable 5.1","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":63.6,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"frontiercode-1-1-extended-score-1-1","evidenceKind":"lab_self_report","harnessId":"frontiercode-1-1-extended-score-1-1:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-189","ingestRunId":"launch-openai-astra-launch","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"FrontierCode 1.1 Extended (score)","locator":"Coding table / FrontierCode 1.1 Extended (score) / Claude Fable 5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":64.9,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"frontiercode-1-1-extended-score-1-1","evidenceKind":"lab_self_report","harnessId":"frontiercode-1-1-extended-score-1-1:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-190","ingestRunId":"launch-openai-astra-launch","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"FrontierCode 1.1 Extended (score)","locator":"Coding table / FrontierCode 1.1 Extended (score) / Claude Opus 5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":63.6,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"frontiercode-1-1-extended-score-1-1","evidenceKind":"lab_self_report","harnessId":"frontiercode-1-1-extended-score-1-1:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-191","ingestRunId":"launch-openai-astra-launch","modelId":"gemini-3.8-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"FrontierCode 1.1 Extended (score)","locator":"Coding table / FrontierCode 1.1 Extended (score) / Gemini 3.8 Flash","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":56.3,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"frontiercode-1-1-main-score-1-1","evidenceKind":"lab_self_report","harnessId":"frontiercode-1-1-main-score-1-1:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IENvZGV4LXN0eWxlIGRldmVsb3BlciBpbnN0cnVjdGlvbiBvbiB0ZXN0cywgcmV1c2UgYW5kIHJlcG9zaXRvcnkgY29udmVudGlvbnM","id":"launch-openai-astra-launch-192","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; Codex-style developer instruction on tests, reuse and repository conventions","family":"FrontierCode 1.1 Main (score)","locator":"Coding table / FrontierCode 1.1 Main (score) / GPT‑6 Astra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":53.3,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"frontiercode-1-1-main-score-1-1","evidenceKind":"lab_self_report","harnessId":"frontiercode-1-1-main-score-1-1:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-193","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"FrontierCode 1.1 Main (score)","locator":"Coding table / FrontierCode 1.1 Main (score) / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":47.5,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"frontiercode-1-1-main-score-1-1","evidenceKind":"lab_self_report","harnessId":"frontiercode-1-1-main-score-1-1:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-194","ingestRunId":"launch-openai-astra-launch","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"FrontierCode 1.1 Main (score)","locator":"Coding table / FrontierCode 1.1 Main (score) / Claude Fable 5.1","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":50.9,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"frontiercode-1-1-main-score-1-1","evidenceKind":"lab_self_report","harnessId":"frontiercode-1-1-main-score-1-1:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-195","ingestRunId":"launch-openai-astra-launch","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"FrontierCode 1.1 Main (score)","locator":"Coding table / FrontierCode 1.1 Main (score) / Claude Fable 5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":53.5,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"frontiercode-1-1-main-score-1-1","evidenceKind":"lab_self_report","harnessId":"frontiercode-1-1-main-score-1-1:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-196","ingestRunId":"launch-openai-astra-launch","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"FrontierCode 1.1 Main (score)","locator":"Coding table / FrontierCode 1.1 Main (score) / Claude Opus 5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":53.4,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"frontiercode-1-1-main-score-1-1","evidenceKind":"lab_self_report","harnessId":"frontiercode-1-1-main-score-1-1:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-197","ingestRunId":"launch-openai-astra-launch","modelId":"gemini-3.8-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"FrontierCode 1.1 Main (score)","locator":"Coding table / FrontierCode 1.1 Main (score) / Gemini 3.8 Flash","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":43.6,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"internal-database-migration-tasks-not-specified","evidenceKind":"lab_self_report","harnessId":"internal-database-migration-tasks-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-198","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"Internal Database Migration Tasks","locator":"Coding table / Internal Database Migration Tasks / GPT‑6 Astra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":63.9,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"internal-database-migration-tasks-not-specified","evidenceKind":"lab_self_report","harnessId":"internal-database-migration-tasks-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-199","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"Internal Database Migration Tasks","locator":"Coding table / Internal Database Migration Tasks / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":42.7,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"internal-database-migration-tasks-not-specified","evidenceKind":"lab_self_report","harnessId":"internal-database-migration-tasks-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-200","ingestRunId":"launch-openai-astra-launch","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"Internal Database Migration Tasks","locator":"Coding table / Internal Database Migration Tasks / Claude Fable 5.1","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":57.8,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"internal-database-migration-tasks-not-specified","evidenceKind":"lab_self_report","harnessId":"internal-database-migration-tasks-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-201","ingestRunId":"launch-openai-astra-launch","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"Internal Database Migration Tasks","locator":"Coding table / Internal Database Migration Tasks / Claude Fable 5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":50.3,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"artificial-analysis-coding-agent-index-v1-4-1-4","evidenceKind":"lab_self_report","harnessId":"artificial-analysis-coding-agent-index-v1-4-1-4:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-202","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"Artificial Analysis Coding Agent Index v1.4","locator":"Coding table / Artificial Analysis Coding Agent Index v1.4 / GPT‑6 Astra","metric":"index score","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":67,"scoreUnit":"index","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"artificial-analysis-coding-agent-index-v1-4-1-4","evidenceKind":"lab_self_report","harnessId":"artificial-analysis-coding-agent-index-v1-4-1-4:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-203","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"Artificial Analysis Coding Agent Index v1.4","locator":"Coding table / Artificial Analysis Coding Agent Index v1.4 / GPT‑5.6 Sol","metric":"index score","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":65.1,"scoreUnit":"index","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"artificial-analysis-coding-agent-index-v1-4-1-4","evidenceKind":"lab_self_report","harnessId":"artificial-analysis-coding-agent-index-v1-4-1-4:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-204","ingestRunId":"launch-openai-astra-launch","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"Artificial Analysis Coding Agent Index v1.4","locator":"Coding table / Artificial Analysis Coding Agent Index v1.4 / Claude Fable 5","metric":"index score","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":67.2,"scoreUnit":"index","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"artificial-analysis-coding-agent-index-v1-4-1-4","evidenceKind":"lab_self_report","harnessId":"artificial-analysis-coding-agent-index-v1-4-1-4:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-205","ingestRunId":"launch-openai-astra-launch","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"Artificial Analysis Coding Agent Index v1.4","locator":"Coding table / Artificial Analysis Coding Agent Index v1.4 / Claude Opus 5","metric":"index score","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":68.1,"scoreUnit":"index","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"artificial-analysis-coding-agent-index-v1-4-1-4","evidenceKind":"lab_self_report","harnessId":"artificial-analysis-coding-agent-index-v1-4-1-4:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-206","ingestRunId":"launch-openai-astra-launch","modelId":"gemini-3.8-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"Artificial Analysis Coding Agent Index v1.4","locator":"Coding table / Artificial Analysis Coding Agent Index v1.4 / Gemini 3.8 Flash","metric":"index score","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":61.2,"scoreUnit":"index","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"terminal-bench-science-0-1-not-specified","evidenceKind":"lab_self_report","harnessId":"terminal-bench-science-0-1-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-207","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"Terminal-Bench Science","locator":"Academic table / Terminal-Bench Science 0.1 / GPT‑6 Astra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":64.6,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"terminal-bench-science-0-1-not-specified","evidenceKind":"lab_self_report","harnessId":"terminal-bench-science-0-1-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-208","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"Terminal-Bench Science","locator":"Academic table / Terminal-Bench Science 0.1 / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":22.4,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"terminal-bench-science-0-1-not-specified","evidenceKind":"lab_self_report","harnessId":"terminal-bench-science-0-1-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-209","ingestRunId":"launch-openai-astra-launch","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"Terminal-Bench Science","locator":"Academic table / Terminal-Bench Science 0.1 / Claude Fable 5.1","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":52.6,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"terminal-bench-science-0-1-not-specified","evidenceKind":"lab_self_report","harnessId":"terminal-bench-science-0-1-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-210","ingestRunId":"launch-openai-astra-launch","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"Terminal-Bench Science","locator":"Academic table / Terminal-Bench Science 0.1 / Claude Fable 5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":21.4,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"terminal-bench-science-0-1-not-specified","evidenceKind":"lab_self_report","harnessId":"terminal-bench-science-0-1-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-211","ingestRunId":"launch-openai-astra-launch","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"Terminal-Bench Science","locator":"Academic table / Terminal-Bench Science 0.1 / Claude Opus 5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":30,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"frontiermath-tier-4-2","evidenceKind":"lab_self_report","harnessId":"frontiermath-tier-4-2:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-212","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"FrontierMath Tier 4 (v2)","locator":"Academic table / FrontierMath Tier 4 (v2) / GPT‑6 Astra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":97.6,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"frontiermath-tier-4-2","evidenceKind":"lab_self_report","harnessId":"frontiermath-tier-4-2:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-213","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"FrontierMath Tier 4 (v2)","locator":"Academic table / FrontierMath Tier 4 (v2) / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":83,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"frontiermath-tier-4-2","evidenceKind":"lab_self_report","harnessId":"frontiermath-tier-4-2:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-214","ingestRunId":"launch-openai-astra-launch","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"FrontierMath Tier 4 (v2)","locator":"Academic table / FrontierMath Tier 4 (v2) / Claude Fable 5.1","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":87.8,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"frontiermath-tier-4-2","evidenceKind":"lab_self_report","harnessId":"frontiermath-tier-4-2:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-215","ingestRunId":"launch-openai-astra-launch","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"FrontierMath Tier 4 (v2)","locator":"Academic table / FrontierMath Tier 4 (v2) / Claude Fable 5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":90.2,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"frontiermath-tier-4-2","evidenceKind":"lab_self_report","harnessId":"frontiermath-tier-4-2:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-216","ingestRunId":"launch-openai-astra-launch","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"FrontierMath Tier 4 (v2)","locator":"Academic table / FrontierMath Tier 4 (v2) / Claude Opus 5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":73.2,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"gpqa-diamond-not-specified","evidenceKind":"lab_self_report","harnessId":"gpqa-diamond-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-217","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"GPQA Diamond","locator":"Academic table / GPQA Diamond / GPT‑6 Astra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":96,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"gpqa-diamond-not-specified","evidenceKind":"lab_self_report","harnessId":"gpqa-diamond-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-218","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"GPQA Diamond","locator":"Academic table / GPQA Diamond / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":94.6,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"gpqa-diamond-not-specified","evidenceKind":"lab_self_report","harnessId":"gpqa-diamond-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-219","ingestRunId":"launch-openai-astra-launch","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"GPQA Diamond","locator":"Academic table / GPQA Diamond / Claude Fable 5.1","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":93.7,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"gpqa-diamond-not-specified","evidenceKind":"lab_self_report","harnessId":"gpqa-diamond-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-220","ingestRunId":"launch-openai-astra-launch","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"GPQA Diamond","locator":"Academic table / GPQA Diamond / Claude Fable 5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":92.6,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"gpqa-diamond-not-specified","evidenceKind":"lab_self_report","harnessId":"gpqa-diamond-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-221","ingestRunId":"launch-openai-astra-launch","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"GPQA Diamond","locator":"Academic table / GPQA Diamond / Claude Opus 5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":93.7,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"gpqa-diamond-not-specified","evidenceKind":"lab_self_report","harnessId":"gpqa-diamond-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-222","ingestRunId":"launch-openai-astra-launch","modelId":"gemini-3.8-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"GPQA Diamond","locator":"Academic table / GPQA Diamond / Gemini 3.8 Flash","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":95.3,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"humanity-s-last-exam-w-tools-not-specified","evidenceKind":"lab_self_report","harnessId":"humanity-s-last-exam-w-tools-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHRvb2xzIGVuYWJsZWQ","id":"launch-openai-astra-launch-223","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; tools enabled","family":"Humanity's Last Exam","locator":"Academic table / Humanity's Last Exam (w/ tools) / GPT‑6 Astra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":57.2,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"humanity-s-last-exam-w-tools-not-specified","evidenceKind":"lab_self_report","harnessId":"humanity-s-last-exam-w-tools-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHRvb2xzIGVuYWJsZWQ","id":"launch-openai-astra-launch-224","ingestRunId":"launch-openai-astra-launch","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; tools enabled","family":"Humanity's Last Exam","locator":"Academic table / Humanity's Last Exam (w/ tools) / Claude Fable 5.1","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":65,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"humanity-s-last-exam-w-tools-not-specified","evidenceKind":"lab_self_report","harnessId":"humanity-s-last-exam-w-tools-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHRvb2xzIGVuYWJsZWQ","id":"launch-openai-astra-launch-225","ingestRunId":"launch-openai-astra-launch","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; tools enabled","family":"Humanity's Last Exam","locator":"Academic table / Humanity's Last Exam (w/ tools) / Claude Fable 5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":63.8,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"humanity-s-last-exam-w-tools-not-specified","evidenceKind":"lab_self_report","harnessId":"humanity-s-last-exam-w-tools-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHRvb2xzIGVuYWJsZWQ","id":"launch-openai-astra-launch-226","ingestRunId":"launch-openai-astra-launch","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; tools enabled","family":"Humanity's Last Exam","locator":"Academic table / Humanity's Last Exam (w/ tools) / Claude Opus 5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":63.6,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"genebench-pro-13","evidenceKind":"lab_self_report","harnessId":"genebench-pro-13:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-227","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"GeneBench Pro","locator":"Science And Health table / GeneBench Pro / GPT‑6 Astra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":37.1,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"genebench-pro-13","evidenceKind":"lab_self_report","harnessId":"genebench-pro-13:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-228","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"GeneBench Pro","locator":"Science And Health table / GeneBench Pro / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":32.3,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"medchembench-internal-not-specified","evidenceKind":"lab_self_report","harnessId":"medchembench-internal-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-229","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"MedChemBench (Internal)","locator":"Science And Health table / MedChemBench (Internal) / GPT‑6 Astra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":49.3,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"medchembench-internal-not-specified","evidenceKind":"lab_self_report","harnessId":"medchembench-internal-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-230","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"MedChemBench (Internal)","locator":"Science And Health table / MedChemBench (Internal) / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":47.4,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"lifescibench-gold-v1","evidenceKind":"lab_self_report","harnessId":"lifescibench-gold-v1:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-231","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"LifeSciBench","locator":"Science And Health table / LifeSciBench / GPT‑6 Astra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":60.3,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"lifescibench-gold-v1","evidenceKind":"lab_self_report","harnessId":"lifescibench-gold-v1:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-232","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"LifeSciBench","locator":"Science And Health table / LifeSciBench / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":59.9,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"healthbench-professional-length-adjusted-not-specified","evidenceKind":"lab_self_report","harnessId":"healthbench-professional-length-adjusted-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IG9mZmljaWFsIHBhcGVyIHNjb3Jpbmc7IGxlbmd0aC1hZGp1c3RlZA","id":"launch-openai-astra-launch-233","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; official paper scoring; length-adjusted","family":"HealthBench Professional (length-adjusted)","locator":"Science And Health table / HealthBench Professional (length-adjusted) / GPT‑6 Astra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":63.4,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"healthbench-professional-length-adjusted-not-specified","evidenceKind":"lab_self_report","harnessId":"healthbench-professional-length-adjusted-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IG9mZmljaWFsIHBhcGVyIHNjb3Jpbmc7IGxlbmd0aC1hZGp1c3RlZA","id":"launch-openai-astra-launch-234","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; official paper scoring; length-adjusted","family":"HealthBench Professional (length-adjusted)","locator":"Science And Health table / HealthBench Professional (length-adjusted) / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":60.5,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"healthbench-professional-length-adjusted-not-specified","evidenceKind":"lab_self_report","harnessId":"healthbench-professional-length-adjusted-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IG9mZmljaWFsIHBhcGVyIHNjb3Jpbmc7IGxlbmd0aC1hZGp1c3RlZDsgT3BlbkFJIHJlcHJvZHVjdGlvbjsgR1BULTUuNCBncmFkZXI7IHVuY2xpcHBlZDsgT3B1czUgZmFsbGJhY2sgZm9yIHByb3ZpZGVyIHJlZnVzYWxz","id":"launch-openai-astra-launch-235","ingestRunId":"launch-openai-astra-launch","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; official paper scoring; length-adjusted; OpenAI reproduction; GPT-5.4 grader; unclipped; Opus5 fallback for provider refusals","family":"HealthBench Professional (length-adjusted)","locator":"Science And Health table / HealthBench Professional (length-adjusted) / Claude Fable 5.1","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":58.1,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"healthbench-professional-length-adjusted-not-specified","evidenceKind":"lab_self_report","harnessId":"healthbench-professional-length-adjusted-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IG9mZmljaWFsIHBhcGVyIHNjb3Jpbmc7IGxlbmd0aC1hZGp1c3RlZDsgT3BlbkFJIHJlcHJvZHVjdGlvbjsgR1BULTUuNCBncmFkZXI7IHVuY2xpcHBlZA","id":"launch-openai-astra-launch-236","ingestRunId":"launch-openai-astra-launch","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; official paper scoring; length-adjusted; OpenAI reproduction; GPT-5.4 grader; unclipped","family":"HealthBench Professional (length-adjusted)","locator":"Science And Health table / HealthBench Professional (length-adjusted) / Claude Fable 5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":60.9,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"healthbench-professional-length-adjusted-not-specified","evidenceKind":"lab_self_report","harnessId":"healthbench-professional-length-adjusted-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IG9mZmljaWFsIHBhcGVyIHNjb3Jpbmc7IGxlbmd0aC1hZGp1c3RlZDsgT3BlbkFJIHJlcHJvZHVjdGlvbjsgR1BULTUuNCBncmFkZXI7IHVuY2xpcHBlZA","id":"launch-openai-astra-launch-237","ingestRunId":"launch-openai-astra-launch","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; official paper scoring; length-adjusted; OpenAI reproduction; GPT-5.4 grader; unclipped","family":"HealthBench Professional (length-adjusted)","locator":"Science And Health table / HealthBench Professional (length-adjusted) / Claude Opus 5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":56.4,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"healthbench-professional-length-adjusted-not-specified","evidenceKind":"lab_self_report","harnessId":"healthbench-professional-length-adjusted-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IG9mZmljaWFsIHBhcGVyIHNjb3Jpbmc7IGxlbmd0aC1hZGp1c3RlZA","id":"launch-openai-astra-launch-238","ingestRunId":"launch-openai-astra-launch","modelId":"gemini-3.8-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; official paper scoring; length-adjusted","family":"HealthBench Professional (length-adjusted)","locator":"Science And Health table / HealthBench Professional (length-adjusted) / Gemini 3.8 Flash","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":52.1,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"exploitbench-not-specified","evidenceKind":"lab_self_report","harnessId":"exploitbench-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHJlZHVjZWQgb3IgYWJzZW50IHByb2R1Y3Rpb24gc2FmZWd1YXJkcw","id":"launch-openai-astra-launch-239","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; reduced or absent production safeguards","family":"ExploitBench","locator":"Cybersecurity table / ExploitBench / GPT‑6 Astra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":100,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"exploitbench-not-specified","evidenceKind":"lab_self_report","harnessId":"exploitbench-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHJlZHVjZWQgb3IgYWJzZW50IHByb2R1Y3Rpb24gc2FmZWd1YXJkcw","id":"launch-openai-astra-launch-240","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; reduced or absent production safeguards","family":"ExploitBench","locator":"Cybersecurity table / ExploitBench / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":78.5,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"exploitbench-not-specified","evidenceKind":"lab_self_report","harnessId":"exploitbench-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHJlZHVjZWQgb3IgYWJzZW50IHByb2R1Y3Rpb24gc2FmZWd1YXJkcw","id":"launch-openai-astra-launch-241","ingestRunId":"launch-openai-astra-launch","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; reduced or absent production safeguards","family":"ExploitBench","locator":"Cybersecurity table / ExploitBench / Claude Opus 5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":70,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"exploitgym-not-specified","evidenceKind":"lab_self_report","harnessId":"exploitgym-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHJlZHVjZWQgb3IgYWJzZW50IHByb2R1Y3Rpb24gc2FmZWd1YXJkczsgdjEgb2ZmbGluZSBlbnZpcm9ubWVudDsgbm8gcnVudGltZSBwYWNrYWdlIGluc3RhbGxhdGlvbjsgdG9rZW4gY2FwcGVkLCBubyB3YWxsLWNsb2NrIGNhcA","id":"launch-openai-astra-launch-242","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; reduced or absent production safeguards; v1 offline environment; no runtime package installation; token capped, no wall-clock cap","family":"ExploitGym","locator":"Cybersecurity table / ExploitGym / GPT‑6 Astra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":42.4,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"exploitgym-not-specified","evidenceKind":"lab_self_report","harnessId":"exploitgym-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHJlZHVjZWQgb3IgYWJzZW50IHByb2R1Y3Rpb24gc2FmZWd1YXJkczsgdjEgb2ZmbGluZSBlbnZpcm9ubWVudDsgbm8gcnVudGltZSBwYWNrYWdlIGluc3RhbGxhdGlvbjsgdG9rZW4gY2FwcGVkLCBubyB3YWxsLWNsb2NrIGNhcA","id":"launch-openai-astra-launch-243","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; reduced or absent production safeguards; v1 offline environment; no runtime package installation; token capped, no wall-clock cap","family":"ExploitGym","locator":"Cybersecurity table / ExploitGym / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":30.3,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"exploitgym-not-specified","evidenceKind":"lab_self_report","harnessId":"exploitgym-not-specified:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHJlZHVjZWQgb3IgYWJzZW50IHByb2R1Y3Rpb24gc2FmZWd1YXJkczsgbGF1bmNoLXJlcG9ydGVkIGNvbmZpZ3VyYXRpb24","id":"launch-openai-astra-launch-244","ingestRunId":"launch-openai-astra-launch","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; reduced or absent production safeguards; launch-reported configuration","family":"ExploitGym","locator":"Cybersecurity table / ExploitGym / Claude Opus 5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":22,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"exploitbench-june-aug-2026-june-aug2026","evidenceKind":"lab_self_report","harnessId":"exploitbench-june-aug-2026-june-aug2026:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHJlZHVjZWQgb3IgYWJzZW50IHByb2R1Y3Rpb24gc2FmZWd1YXJkczsgMjAgdnVsbmVyYWJpbGl0aWVzIC8gMTMgQ2hyb21lIHJlbGVhc2Vz","id":"launch-openai-astra-launch-245","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; reduced or absent production safeguards; 20 vulnerabilities / 13 Chrome releases","family":"ExploitBench (June-Aug 2026)","locator":"Cybersecurity table / ExploitBench (June-Aug 2026) / GPT‑6 Astra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":39,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"exploitbench-june-aug-2026-june-aug2026","evidenceKind":"lab_self_report","harnessId":"exploitbench-june-aug-2026-june-aug2026:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHJlZHVjZWQgb3IgYWJzZW50IHByb2R1Y3Rpb24gc2FmZWd1YXJkczsgMjAgdnVsbmVyYWJpbGl0aWVzIC8gMTMgQ2hyb21lIHJlbGVhc2VzOyAzMDAtdHVybiBsaW1pdA","id":"launch-openai-astra-launch-246","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; reduced or absent production safeguards; 20 vulnerabilities / 13 Chrome releases; 300-turn limit","family":"ExploitBench (June-Aug 2026)","locator":"Cybersecurity table / ExploitBench (June-Aug 2026) / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced. Footnote14 separately reports11.5% with fewer turn-limit interruptions; main table remains5.5%.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":5.5,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"sre-bench-262-binaries-19-programs","evidenceKind":"lab_self_report","harnessId":"sre-bench-262-binaries-19-programs:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHJlZHVjZWQgb3IgYWJzZW50IHByb2R1Y3Rpb24gc2FmZWd1YXJkczsgcGFzc0AxOyBhbGwgc2l4IG9iamVjdGl2ZXMgcmVxdWlyZWQ","id":"launch-openai-astra-launch-247","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; reduced or absent production safeguards; pass@1; all six objectives required","family":"SRE-Bench","locator":"Cybersecurity table / SRE-Bench / GPT‑6 Astra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":88,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"sre-bench-262-binaries-19-programs","evidenceKind":"lab_self_report","harnessId":"sre-bench-262-binaries-19-programs:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHJlZHVjZWQgb3IgYWJzZW50IHByb2R1Y3Rpb24gc2FmZWd1YXJkczsgcGFzc0AxOyBhbGwgc2l4IG9iamVjdGl2ZXMgcmVxdWlyZWQ","id":"launch-openai-astra-launch-248","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; reduced or absent production safeguards; pass@1; all six objectives required","family":"SRE-Bench","locator":"Cybersecurity table / SRE-Bench / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":55.9,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"sre-bench-262-binaries-19-programs","evidenceKind":"lab_self_report","harnessId":"sre-bench-262-binaries-19-programs:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHJlZHVjZWQgb3IgYWJzZW50IHByb2R1Y3Rpb24gc2FmZWd1YXJkczsgcGFzc0AxOyBhbGwgc2l4IG9iamVjdGl2ZXMgcmVxdWlyZWQ","id":"launch-openai-astra-launch-249","ingestRunId":"launch-openai-astra-launch","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; reduced or absent production safeguards; pass@1; all six objectives required","family":"SRE-Bench","locator":"Cybersecurity table / SRE-Bench / Claude Opus 5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":12.5,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"sec-bench-pro-may2026-revised-root-cause-grader","evidenceKind":"lab_self_report","harnessId":"sec-bench-pro-may2026-revised-root-cause-grader:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHJlZHVjZWQgb3IgYWJzZW50IHByb2R1Y3Rpb24gc2FmZWd1YXJkczsgTWF5MjAyNiBKYXZhU2NyaXB0IHN1YnNldCwxODMgdnVsbmVyYWJpbGl0aWVzOyByZXZpc2VkIGFnZW50IHJvb3QtY2F1c2UgZ3JhZGVy","id":"launch-openai-astra-launch-250","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; reduced or absent production safeguards; May2026 JavaScript subset,183 vulnerabilities; revised agent root-cause grader","family":"SEC-Bench Pro","locator":"Cybersecurity table / SEC-Bench Pro / GPT‑6 Astra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":85.4,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"sec-bench-pro-may2026-revised-root-cause-grader","evidenceKind":"lab_self_report","harnessId":"sec-bench-pro-may2026-revised-root-cause-grader:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHJlZHVjZWQgb3IgYWJzZW50IHByb2R1Y3Rpb24gc2FmZWd1YXJkczsgTWF5MjAyNiBKYXZhU2NyaXB0IHN1YnNldCwxODMgdnVsbmVyYWJpbGl0aWVzOyByZXZpc2VkIGFnZW50IHJvb3QtY2F1c2UgZ3JhZGVy","id":"launch-openai-astra-launch-251","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; reduced or absent production safeguards; May2026 JavaScript subset,183 vulnerabilities; revised agent root-cause grader","family":"SEC-Bench Pro","locator":"Cybersecurity table / SEC-Bench Pro / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":79.1,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"openai-mrcr-v2-8-needle-256k-512k-2","evidenceKind":"lab_self_report","harnessId":"openai-mrcr-v2-8-needle-256k-512k-2:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IDgtbmVlZGxlIDI1NkstNTEySw","id":"launch-openai-astra-launch-252","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; 8-needle 256K-512K","family":"OpenAI MRCR v2 8-needle 256K-512K","locator":"Long Context table / OpenAI MRCR v2 8-needle 256K-512K / GPT‑6 Astra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":100,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"openai-mrcr-v2-8-needle-256k-512k-2","evidenceKind":"lab_self_report","harnessId":"openai-mrcr-v2-8-needle-256k-512k-2:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IDgtbmVlZGxlIDI1NkstNTEySw","id":"launch-openai-astra-launch-253","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; 8-needle 256K-512K","family":"OpenAI MRCR v2 8-needle 256K-512K","locator":"Long Context table / OpenAI MRCR v2 8-needle 256K-512K / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":91.5,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"openai-mrcr-v2-8-needle-512k-1m-2","evidenceKind":"lab_self_report","harnessId":"openai-mrcr-v2-8-needle-512k-1m-2:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IDgtbmVlZGxlIDUxMkstMU0","id":"launch-openai-astra-launch-254","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; 8-needle 512K-1M","family":"OpenAI MRCR v2 8-needle 512K-1M","locator":"Long Context table / OpenAI MRCR v2 8-needle 512K-1M / GPT‑6 Astra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":96.3,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"openai-mrcr-v2-8-needle-512k-1m-2","evidenceKind":"lab_self_report","harnessId":"openai-mrcr-v2-8-needle-512k-1m-2:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IDgtbmVlZGxlIDUxMkstMU0","id":"launch-openai-astra-launch-255","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; 8-needle 512K-1M","family":"OpenAI MRCR v2 8-needle 512K-1M","locator":"Long Context table / OpenAI MRCR v2 8-needle 512K-1M / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":73.8,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"arc-agi-3","evidenceKind":"lab_self_report","harnessId":"arc-agi-3:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IFJlc3BvbnNlcyBBUEkgaGFybmVzcyB3aXRoIHR3byBkb2N1bWVudGVkIHNldHRpbmcgY2hhbmdlcw","id":"launch-openai-astra-launch-256","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; Responses API harness with two documented setting changes","family":"ARC-AGI-3","locator":"Abstract Reasoning table / ARC-AGI-3 / GPT‑6 Astra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":99.9,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"arc-agi-3","evidenceKind":"lab_self_report","harnessId":"arc-agi-3:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-257","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"ARC-AGI-3","locator":"Abstract Reasoning table / ARC-AGI-3 / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":7.8,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"arc-agi-3","evidenceKind":"lab_self_report","harnessId":"arc-agi-3:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-258","ingestRunId":"launch-openai-astra-launch","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"ARC-AGI-3","locator":"Abstract Reasoning table / ARC-AGI-3 / Claude Opus 5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":30.2,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"arc-agi-2-percent","evidenceKind":"lab_self_report","harnessId":"arc-agi-2-percent:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-259","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"ARC-AGI-2","locator":"Abstract Reasoning table / ARC-AGI-2 / GPT‑6 Astra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":95,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"arc-agi-2-percent","evidenceKind":"lab_self_report","harnessId":"arc-agi-2-percent:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-260","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"ARC-AGI-2","locator":"Abstract Reasoning table / ARC-AGI-2 / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":92.5,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"arc-agi-2-percent","evidenceKind":"lab_self_report","harnessId":"arc-agi-2-percent:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-261","ingestRunId":"launch-openai-astra-launch","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"ARC-AGI-2","locator":"Abstract Reasoning table / ARC-AGI-2 / Claude Fable 5.1","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":90,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"arc-agi-2-percent","evidenceKind":"lab_self_report","harnessId":"arc-agi-2-percent:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-262","ingestRunId":"launch-openai-astra-launch","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"ARC-AGI-2","locator":"Abstract Reasoning table / ARC-AGI-2 / Claude Fable 5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":89.2,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"arc-agi-2-percent","evidenceKind":"lab_self_report","harnessId":"arc-agi-2-percent:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-263","ingestRunId":"launch-openai-astra-launch","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"ARC-AGI-2","locator":"Abstract Reasoning table / ARC-AGI-2 / Claude Opus 5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":90.4,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"arc-agi-1","evidenceKind":"lab_self_report","harnessId":"arc-agi-1:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-264","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"ARC-AGI-1","locator":"Abstract Reasoning table / ARC-AGI-1 / GPT‑6 Astra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":98.5,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"arc-agi-1","evidenceKind":"lab_self_report","harnessId":"arc-agi-1:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-265","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"ARC-AGI-1","locator":"Abstract Reasoning table / ARC-AGI-1 / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":97.5,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"arc-agi-1","evidenceKind":"lab_self_report","harnessId":"arc-agi-1:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-266","ingestRunId":"launch-openai-astra-launch","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"ARC-AGI-1","locator":"Abstract Reasoning table / ARC-AGI-1 / Claude Fable 5.1","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":97.5,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"arc-agi-1","evidenceKind":"lab_self_report","harnessId":"arc-agi-1:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-267","ingestRunId":"launch-openai-astra-launch","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"ARC-AGI-1","locator":"Abstract Reasoning table / ARC-AGI-1 / Claude Fable 5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":98.5,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"arc-agi-1","evidenceKind":"lab_self_report","harnessId":"arc-agi-1:openai-astra-launch:TWF4aW11bSByZXBvcnRlZCBhY3Jvc3MgcmVhc29uaW5nIGVmZm9ydHM7IE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk","id":"launch-openai-astra-launch-268","ingestRunId":"launch-openai-astra-launch","modelId":"claude-opus-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API","family":"ARC-AGI-1","locator":"Abstract Reasoning table / ARC-AGI-1 / Claude Opus 5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":97.5,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"terminal-bench-science-0-1","evidenceKind":"lab_self_report","harnessId":"terminal-bench-science-0-1:openai-astra-launch:TG93ZXItY29zdCBzZXR0aW5nOyBleGFjdCBlZmZvcnQgbm90IHNwZWNpZmllZA","id":"launch-openai-astra-launch-547","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Lower-cost setting; exact effort not specified","family":"Terminal-Bench Science","locator":"Terminal-Bench Science chart caption","metric":"percent","notes":"Provider-published result; preserves source setting without claiming a controlled cross-provider comparison.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":61.1,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"gpqa-diamond-not-specified","evidenceKind":"lab_self_report","harnessId":"gpqa-diamond-not-specified:openai-astra-launch:TG93ZXItY29zdCBzZXR0aW5nOyBleGFjdCBlZmZvcnQgbm90IHNwZWNpZmllZA","id":"launch-openai-astra-launch-548","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Lower-cost setting; exact effort not specified","family":"GPQA Diamond","locator":"GPQA Diamond chart caption","metric":"percent","notes":"Provider-published result; preserves source setting without claiming a controlled cross-provider comparison.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":94.9,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"exploitbench-june-aug-2026-june-aug2026","evidenceKind":"lab_self_report","harnessId":"exploitbench-june-aug-2026-june-aug2026:openai-astra-launch:U2ltaWxhciBzZXR0aW5ncyB3aXRoIGZld2VyMzAwLXR1cm4tbGltaXQgaW50ZXJydXB0aW9uczsgcHJvZHVjdGlvbiBzYWZlZ3VhcmRzIGFic2VudA","id":"launch-openai-astra-launch-549","ingestRunId":"launch-openai-astra-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Similar settings with fewer300-turn-limit interruptions; production safeguards absent","family":"ExploitBench (June-Aug 2026)","locator":"Footnote14","metric":"percent","notes":"Provider-published result; preserves source setting without claiming a controlled cross-provider comparison.","sourceId":"openai-astra-launch","sourceTitle":"GPT-6 Astra: A new generation of intelligence"},"score":11.5,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-6-astra/"},{"benchmarkId":"healthbench-professional-not-specified-score-0-100","evidenceKind":"lab_self_report","harnessId":"healthbench-professional-not-specified-score-0-100:openai-astra-system-card:bGVuZ3RoLWFkanVzdGVkOyBvZmZpY2lhbCBIZWFsdGhCZW5jaCBzY29yaW5n","id":"launch-openai-astra-system-card-489","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"length-adjusted; official HealthBench scoring","family":"HealthBench Professional","locator":"Table6 / HealthBench Professional / gpt-5.5 / length-adjusted","metric":"score (0-100)","notes":"Mean response length 3818 characters. Adjustment centered on2000characters; adjusted and unadjusted values remain distinct configurations.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":51.8,"scoreUnit":"index","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"healthbench-professional-not-specified-score-0-100","evidenceKind":"lab_self_report","harnessId":"healthbench-professional-not-specified-score-0-100:openai-astra-system-card:dW5hZGp1c3RlZDsgb2ZmaWNpYWwgSGVhbHRoQmVuY2ggc2NvcmluZw","id":"launch-openai-astra-system-card-490","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"unadjusted; official HealthBench scoring","family":"HealthBench Professional","locator":"Table6 / HealthBench Professional / gpt-5.5 / unadjusted","metric":"score (0-100)","notes":"Mean response length 3818 characters. Adjustment centered on2000characters; adjusted and unadjusted values remain distinct configurations.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":57.2,"scoreUnit":"index","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"healthbench-professional-not-specified-score-0-100","evidenceKind":"lab_self_report","harnessId":"healthbench-professional-not-specified-score-0-100:openai-astra-system-card:bGVuZ3RoLWFkanVzdGVkOyBvZmZpY2lhbCBIZWFsdGhCZW5jaCBzY29yaW5n","id":"launch-openai-astra-system-card-491","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"length-adjusted; official HealthBench scoring","family":"HealthBench Professional","locator":"Table6 / HealthBench Professional / gpt-5.6-sol / length-adjusted","metric":"score (0-100)","notes":"Mean response length 3228 characters. Adjustment centered on2000characters; adjusted and unadjusted values remain distinct configurations.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":60.5,"scoreUnit":"index","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"healthbench-professional-not-specified-score-0-100","evidenceKind":"lab_self_report","harnessId":"healthbench-professional-not-specified-score-0-100:openai-astra-system-card:dW5hZGp1c3RlZDsgb2ZmaWNpYWwgSGVhbHRoQmVuY2ggc2NvcmluZw","id":"launch-openai-astra-system-card-492","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"unadjusted; official HealthBench scoring","family":"HealthBench Professional","locator":"Table6 / HealthBench Professional / gpt-5.6-sol / unadjusted","metric":"score (0-100)","notes":"Mean response length 3228 characters. Adjustment centered on2000characters; adjusted and unadjusted values remain distinct configurations.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":64.1,"scoreUnit":"index","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"healthbench-professional-not-specified-score-0-100","evidenceKind":"lab_self_report","harnessId":"healthbench-professional-not-specified-score-0-100:openai-astra-system-card:bGVuZ3RoLWFkanVzdGVkOyBvZmZpY2lhbCBIZWFsdGhCZW5jaCBzY29yaW5n","id":"launch-openai-astra-system-card-493","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"length-adjusted; official HealthBench scoring","family":"HealthBench Professional","locator":"Table6 / HealthBench Professional / gpt-5.6-terra / length-adjusted","metric":"score (0-100)","notes":"Mean response length 3618 characters. Adjustment centered on2000characters; adjusted and unadjusted values remain distinct configurations.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":57.7,"scoreUnit":"index","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"healthbench-professional-not-specified-score-0-100","evidenceKind":"lab_self_report","harnessId":"healthbench-professional-not-specified-score-0-100:openai-astra-system-card:dW5hZGp1c3RlZDsgb2ZmaWNpYWwgSGVhbHRoQmVuY2ggc2NvcmluZw","id":"launch-openai-astra-system-card-494","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"unadjusted; official HealthBench scoring","family":"HealthBench Professional","locator":"Table6 / HealthBench Professional / gpt-5.6-terra / unadjusted","metric":"score (0-100)","notes":"Mean response length 3618 characters. Adjustment centered on2000characters; adjusted and unadjusted values remain distinct configurations.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":62.4,"scoreUnit":"index","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"healthbench-professional-not-specified-score-0-100","evidenceKind":"lab_self_report","harnessId":"healthbench-professional-not-specified-score-0-100:openai-astra-system-card:bGVuZ3RoLWFkanVzdGVkOyBvZmZpY2lhbCBIZWFsdGhCZW5jaCBzY29yaW5n","id":"launch-openai-astra-system-card-495","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-5.6-luna","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"length-adjusted; official HealthBench scoring","family":"HealthBench Professional","locator":"Table6 / HealthBench Professional / gpt-5.6-luna / length-adjusted","metric":"score (0-100)","notes":"Mean response length 3389 characters. Adjustment centered on2000characters; adjusted and unadjusted values remain distinct configurations.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":55.7,"scoreUnit":"index","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"healthbench-professional-not-specified-score-0-100","evidenceKind":"lab_self_report","harnessId":"healthbench-professional-not-specified-score-0-100:openai-astra-system-card:dW5hZGp1c3RlZDsgb2ZmaWNpYWwgSGVhbHRoQmVuY2ggc2NvcmluZw","id":"launch-openai-astra-system-card-496","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-5.6-luna","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"unadjusted; official HealthBench scoring","family":"HealthBench Professional","locator":"Table6 / HealthBench Professional / gpt-5.6-luna / unadjusted","metric":"score (0-100)","notes":"Mean response length 3389 characters. Adjustment centered on2000characters; adjusted and unadjusted values remain distinct configurations.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":59.8,"scoreUnit":"index","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"healthbench-professional-not-specified-score-0-100","evidenceKind":"lab_self_report","harnessId":"healthbench-professional-not-specified-score-0-100:openai-astra-system-card:bGVuZ3RoLWFkanVzdGVkOyBvZmZpY2lhbCBIZWFsdGhCZW5jaCBzY29yaW5n","id":"launch-openai-astra-system-card-497","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"length-adjusted; official HealthBench scoring","family":"HealthBench Professional","locator":"Table6 / HealthBench Professional / gpt-6-astra / length-adjusted","metric":"score (0-100)","notes":"Mean response length 4097 characters. Adjustment centered on2000characters; adjusted and unadjusted values remain distinct configurations.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":63.4,"scoreUnit":"index","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"healthbench-professional-not-specified-score-0-100","evidenceKind":"lab_self_report","harnessId":"healthbench-professional-not-specified-score-0-100:openai-astra-system-card:dW5hZGp1c3RlZDsgb2ZmaWNpYWwgSGVhbHRoQmVuY2ggc2NvcmluZw","id":"launch-openai-astra-system-card-498","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"unadjusted; official HealthBench scoring","family":"HealthBench Professional","locator":"Table6 / HealthBench Professional / gpt-6-astra / unadjusted","metric":"score (0-100)","notes":"Mean response length 4097 characters. Adjustment centered on2000characters; adjusted and unadjusted values remain distinct configurations.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":69.5,"scoreUnit":"index","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"healthbench-not-specified","evidenceKind":"lab_self_report","harnessId":"healthbench-not-specified:openai-astra-system-card:bGVuZ3RoLWFkanVzdGVkOyBvZmZpY2lhbCBIZWFsdGhCZW5jaCBzY29yaW5n","id":"launch-openai-astra-system-card-499","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"length-adjusted; official HealthBench scoring","family":"HealthBench","locator":"Table6 / HealthBench / gpt-5.5 / length-adjusted","metric":"score (0-100)","notes":"Mean response length 2313 characters. Adjustment centered on2000characters; adjusted and unadjusted values remain distinct configurations.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":56.5,"scoreUnit":"index","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"healthbench-not-specified","evidenceKind":"lab_self_report","harnessId":"healthbench-not-specified:openai-astra-system-card:dW5hZGp1c3RlZDsgb2ZmaWNpYWwgSGVhbHRoQmVuY2ggc2NvcmluZw","id":"launch-openai-astra-system-card-500","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"unadjusted; official HealthBench scoring","family":"HealthBench","locator":"Table6 / HealthBench / gpt-5.5 / unadjusted","metric":"score (0-100)","notes":"Mean response length 2313 characters. Adjustment centered on2000characters; adjusted and unadjusted values remain distinct configurations.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":58.4,"scoreUnit":"index","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"healthbench-not-specified","evidenceKind":"lab_self_report","harnessId":"healthbench-not-specified:openai-astra-system-card:bGVuZ3RoLWFkanVzdGVkOyBvZmZpY2lhbCBIZWFsdGhCZW5jaCBzY29yaW5n","id":"launch-openai-astra-system-card-501","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"length-adjusted; official HealthBench scoring","family":"HealthBench","locator":"Table6 / HealthBench / gpt-5.6-sol / length-adjusted","metric":"score (0-100)","notes":"Mean response length 1764 characters. Adjustment centered on2000characters; adjusted and unadjusted values remain distinct configurations.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":57,"scoreUnit":"index","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"healthbench-not-specified","evidenceKind":"lab_self_report","harnessId":"healthbench-not-specified:openai-astra-system-card:dW5hZGp1c3RlZDsgb2ZmaWNpYWwgSGVhbHRoQmVuY2ggc2NvcmluZw","id":"launch-openai-astra-system-card-502","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"unadjusted; official HealthBench scoring","family":"HealthBench","locator":"Table6 / HealthBench / gpt-5.6-sol / unadjusted","metric":"score (0-100)","notes":"Mean response length 1764 characters. Adjustment centered on2000characters; adjusted and unadjusted values remain distinct configurations.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":55.6,"scoreUnit":"index","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"healthbench-not-specified","evidenceKind":"lab_self_report","harnessId":"healthbench-not-specified:openai-astra-system-card:bGVuZ3RoLWFkanVzdGVkOyBvZmZpY2lhbCBIZWFsdGhCZW5jaCBzY29yaW5n","id":"launch-openai-astra-system-card-503","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"length-adjusted; official HealthBench scoring","family":"HealthBench","locator":"Table6 / HealthBench / gpt-5.6-terra / length-adjusted","metric":"score (0-100)","notes":"Mean response length 2285 characters. Adjustment centered on2000characters; adjusted and unadjusted values remain distinct configurations.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":57,"scoreUnit":"index","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"healthbench-not-specified","evidenceKind":"lab_self_report","harnessId":"healthbench-not-specified:openai-astra-system-card:dW5hZGp1c3RlZDsgb2ZmaWNpYWwgSGVhbHRoQmVuY2ggc2NvcmluZw","id":"launch-openai-astra-system-card-504","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"unadjusted; official HealthBench scoring","family":"HealthBench","locator":"Table6 / HealthBench / gpt-5.6-terra / unadjusted","metric":"score (0-100)","notes":"Mean response length 2285 characters. Adjustment centered on2000characters; adjusted and unadjusted values remain distinct configurations.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":58.7,"scoreUnit":"index","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"healthbench-not-specified","evidenceKind":"lab_self_report","harnessId":"healthbench-not-specified:openai-astra-system-card:bGVuZ3RoLWFkanVzdGVkOyBvZmZpY2lhbCBIZWFsdGhCZW5jaCBzY29yaW5n","id":"launch-openai-astra-system-card-505","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-5.6-luna","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"length-adjusted; official HealthBench scoring","family":"HealthBench","locator":"Table6 / HealthBench / gpt-5.6-luna / length-adjusted","metric":"score (0-100)","notes":"Mean response length 1930 characters. Adjustment centered on2000characters; adjusted and unadjusted values remain distinct configurations.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":55.8,"scoreUnit":"index","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"healthbench-not-specified","evidenceKind":"lab_self_report","harnessId":"healthbench-not-specified:openai-astra-system-card:dW5hZGp1c3RlZDsgb2ZmaWNpYWwgSGVhbHRoQmVuY2ggc2NvcmluZw","id":"launch-openai-astra-system-card-506","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-5.6-luna","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"unadjusted; official HealthBench scoring","family":"HealthBench","locator":"Table6 / HealthBench / gpt-5.6-luna / unadjusted","metric":"score (0-100)","notes":"Mean response length 1930 characters. Adjustment centered on2000characters; adjusted and unadjusted values remain distinct configurations.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":55.4,"scoreUnit":"index","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"healthbench-not-specified","evidenceKind":"lab_self_report","harnessId":"healthbench-not-specified:openai-astra-system-card:bGVuZ3RoLWFkanVzdGVkOyBvZmZpY2lhbCBIZWFsdGhCZW5jaCBzY29yaW5n","id":"launch-openai-astra-system-card-507","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"length-adjusted; official HealthBench scoring","family":"HealthBench","locator":"Table6 / HealthBench / gpt-6-astra / length-adjusted","metric":"score (0-100)","notes":"Mean response length 2258 characters. Adjustment centered on2000characters; adjusted and unadjusted values remain distinct configurations.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":58.1,"scoreUnit":"index","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"healthbench-not-specified","evidenceKind":"lab_self_report","harnessId":"healthbench-not-specified:openai-astra-system-card:dW5hZGp1c3RlZDsgb2ZmaWNpYWwgSGVhbHRoQmVuY2ggc2NvcmluZw","id":"launch-openai-astra-system-card-508","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"unadjusted; official HealthBench scoring","family":"HealthBench","locator":"Table6 / HealthBench / gpt-6-astra / unadjusted","metric":"score (0-100)","notes":"Mean response length 2258 characters. Adjustment centered on2000characters; adjusted and unadjusted values remain distinct configurations.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":59.7,"scoreUnit":"index","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"healthbench-hard-not-specified","evidenceKind":"lab_self_report","harnessId":"healthbench-hard-not-specified:openai-astra-system-card:bGVuZ3RoLWFkanVzdGVkOyBvZmZpY2lhbCBIZWFsdGhCZW5jaCBzY29yaW5n","id":"launch-openai-astra-system-card-509","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"length-adjusted; official HealthBench scoring","family":"HealthBench Hard","locator":"Table6 / HealthBench Hard / gpt-5.5 / length-adjusted","metric":"score (0-100)","notes":"Mean response length 2289 characters. Adjustment centered on2000characters; adjusted and unadjusted values remain distinct configurations.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":31.5,"scoreUnit":"index","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"healthbench-hard-not-specified","evidenceKind":"lab_self_report","harnessId":"healthbench-hard-not-specified:openai-astra-system-card:dW5hZGp1c3RlZDsgb2ZmaWNpYWwgSGVhbHRoQmVuY2ggc2NvcmluZw","id":"launch-openai-astra-system-card-510","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"unadjusted; official HealthBench scoring","family":"HealthBench Hard","locator":"Table6 / HealthBench Hard / gpt-5.5 / unadjusted","metric":"score (0-100)","notes":"Mean response length 2289 characters. Adjustment centered on2000characters; adjusted and unadjusted values remain distinct configurations.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":33.8,"scoreUnit":"index","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"healthbench-hard-not-specified","evidenceKind":"lab_self_report","harnessId":"healthbench-hard-not-specified:openai-astra-system-card:bGVuZ3RoLWFkanVzdGVkOyBvZmZpY2lhbCBIZWFsdGhCZW5jaCBzY29yaW5n","id":"launch-openai-astra-system-card-511","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"length-adjusted; official HealthBench scoring","family":"HealthBench Hard","locator":"Table6 / HealthBench Hard / gpt-5.6-sol / length-adjusted","metric":"score (0-100)","notes":"Mean response length 1751 characters. Adjustment centered on2000characters; adjusted and unadjusted values remain distinct configurations.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":33.1,"scoreUnit":"index","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"healthbench-hard-not-specified","evidenceKind":"lab_self_report","harnessId":"healthbench-hard-not-specified:openai-astra-system-card:dW5hZGp1c3RlZDsgb2ZmaWNpYWwgSGVhbHRoQmVuY2ggc2NvcmluZw","id":"launch-openai-astra-system-card-512","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"unadjusted; official HealthBench scoring","family":"HealthBench Hard","locator":"Table6 / HealthBench Hard / gpt-5.6-sol / unadjusted","metric":"score (0-100)","notes":"Mean response length 1751 characters. Adjustment centered on2000characters; adjusted and unadjusted values remain distinct configurations.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":31.1,"scoreUnit":"index","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"healthbench-hard-not-specified","evidenceKind":"lab_self_report","harnessId":"healthbench-hard-not-specified:openai-astra-system-card:bGVuZ3RoLWFkanVzdGVkOyBvZmZpY2lhbCBIZWFsdGhCZW5jaCBzY29yaW5n","id":"launch-openai-astra-system-card-513","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"length-adjusted; official HealthBench scoring","family":"HealthBench Hard","locator":"Table6 / HealthBench Hard / gpt-5.6-terra / length-adjusted","metric":"score (0-100)","notes":"Mean response length 2199 characters. Adjustment centered on2000characters; adjusted and unadjusted values remain distinct configurations.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":32.7,"scoreUnit":"index","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"healthbench-hard-not-specified","evidenceKind":"lab_self_report","harnessId":"healthbench-hard-not-specified:openai-astra-system-card:dW5hZGp1c3RlZDsgb2ZmaWNpYWwgSGVhbHRoQmVuY2ggc2NvcmluZw","id":"launch-openai-astra-system-card-514","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"unadjusted; official HealthBench scoring","family":"HealthBench Hard","locator":"Table6 / HealthBench Hard / gpt-5.6-terra / unadjusted","metric":"score (0-100)","notes":"Mean response length 2199 characters. Adjustment centered on2000characters; adjusted and unadjusted values remain distinct configurations.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":34.3,"scoreUnit":"index","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"healthbench-hard-not-specified","evidenceKind":"lab_self_report","harnessId":"healthbench-hard-not-specified:openai-astra-system-card:bGVuZ3RoLWFkanVzdGVkOyBvZmZpY2lhbCBIZWFsdGhCZW5jaCBzY29yaW5n","id":"launch-openai-astra-system-card-515","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-5.6-luna","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"length-adjusted; official HealthBench scoring","family":"HealthBench Hard","locator":"Table6 / HealthBench Hard / gpt-5.6-luna / length-adjusted","metric":"score (0-100)","notes":"Mean response length 1923 characters. Adjustment centered on2000characters; adjusted and unadjusted values remain distinct configurations.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":32,"scoreUnit":"index","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"healthbench-hard-not-specified","evidenceKind":"lab_self_report","harnessId":"healthbench-hard-not-specified:openai-astra-system-card:dW5hZGp1c3RlZDsgb2ZmaWNpYWwgSGVhbHRoQmVuY2ggc2NvcmluZw","id":"launch-openai-astra-system-card-516","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-5.6-luna","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"unadjusted; official HealthBench scoring","family":"HealthBench Hard","locator":"Table6 / HealthBench Hard / gpt-5.6-luna / unadjusted","metric":"score (0-100)","notes":"Mean response length 1923 characters. Adjustment centered on2000characters; adjusted and unadjusted values remain distinct configurations.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":31.4,"scoreUnit":"index","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"healthbench-hard-not-specified","evidenceKind":"lab_self_report","harnessId":"healthbench-hard-not-specified:openai-astra-system-card:bGVuZ3RoLWFkanVzdGVkOyBvZmZpY2lhbCBIZWFsdGhCZW5jaCBzY29yaW5n","id":"launch-openai-astra-system-card-517","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"length-adjusted; official HealthBench scoring","family":"HealthBench Hard","locator":"Table6 / HealthBench Hard / gpt-6-astra / length-adjusted","metric":"score (0-100)","notes":"Mean response length 2192 characters. Adjustment centered on2000characters; adjusted and unadjusted values remain distinct configurations.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":36.3,"scoreUnit":"index","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"healthbench-hard-not-specified","evidenceKind":"lab_self_report","harnessId":"healthbench-hard-not-specified:openai-astra-system-card:dW5hZGp1c3RlZDsgb2ZmaWNpYWwgSGVhbHRoQmVuY2ggc2NvcmluZw","id":"launch-openai-astra-system-card-518","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"unadjusted; official HealthBench scoring","family":"HealthBench Hard","locator":"Table6 / HealthBench Hard / gpt-6-astra / unadjusted","metric":"score (0-100)","notes":"Mean response length 2192 characters. Adjustment centered on2000characters; adjusted and unadjusted values remain distinct configurations.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":37.8,"scoreUnit":"index","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"healthbench-consensus-not-specified","evidenceKind":"lab_self_report","harnessId":"healthbench-consensus-not-specified:openai-astra-system-card:bGVuZ3RoLWFkanVzdGVkOyBvZmZpY2lhbCBIZWFsdGhCZW5jaCBzY29yaW5n","id":"launch-openai-astra-system-card-519","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"length-adjusted; official HealthBench scoring","family":"HealthBench Consensus","locator":"Table6 / HealthBench Consensus / gpt-5.5 / length-adjusted","metric":"score (0-100)","notes":"Mean response length 2259 characters. Adjustment centered on2000characters; adjusted and unadjusted values remain distinct configurations.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":95.6,"scoreUnit":"index","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"healthbench-consensus-not-specified","evidenceKind":"lab_self_report","harnessId":"healthbench-consensus-not-specified:openai-astra-system-card:dW5hZGp1c3RlZDsgb2ZmaWNpYWwgSGVhbHRoQmVuY2ggc2NvcmluZw","id":"launch-openai-astra-system-card-520","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"unadjusted; official HealthBench scoring","family":"HealthBench Consensus","locator":"Table6 / HealthBench Consensus / gpt-5.5 / unadjusted","metric":"score (0-100)","notes":"Mean response length 2259 characters. Adjustment centered on2000characters; adjusted and unadjusted values remain distinct configurations.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":95.7,"scoreUnit":"index","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"healthbench-consensus-not-specified","evidenceKind":"lab_self_report","harnessId":"healthbench-consensus-not-specified:openai-astra-system-card:bGVuZ3RoLWFkanVzdGVkOyBvZmZpY2lhbCBIZWFsdGhCZW5jaCBzY29yaW5n","id":"launch-openai-astra-system-card-521","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"length-adjusted; official HealthBench scoring","family":"HealthBench Consensus","locator":"Table6 / HealthBench Consensus / gpt-5.6-sol / length-adjusted","metric":"score (0-100)","notes":"Mean response length 1740 characters. Adjustment centered on2000characters; adjusted and unadjusted values remain distinct configurations.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":95.5,"scoreUnit":"index","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"healthbench-consensus-not-specified","evidenceKind":"lab_self_report","harnessId":"healthbench-consensus-not-specified:openai-astra-system-card:dW5hZGp1c3RlZDsgb2ZmaWNpYWwgSGVhbHRoQmVuY2ggc2NvcmluZw","id":"launch-openai-astra-system-card-522","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"unadjusted; official HealthBench scoring","family":"HealthBench Consensus","locator":"Table6 / HealthBench Consensus / gpt-5.6-sol / unadjusted","metric":"score (0-100)","notes":"Mean response length 1740 characters. Adjustment centered on2000characters; adjusted and unadjusted values remain distinct configurations.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":95.3,"scoreUnit":"index","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"healthbench-consensus-not-specified","evidenceKind":"lab_self_report","harnessId":"healthbench-consensus-not-specified:openai-astra-system-card:bGVuZ3RoLWFkanVzdGVkOyBvZmZpY2lhbCBIZWFsdGhCZW5jaCBzY29yaW5n","id":"launch-openai-astra-system-card-523","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"length-adjusted; official HealthBench scoring","family":"HealthBench Consensus","locator":"Table6 / HealthBench Consensus / gpt-5.6-terra / length-adjusted","metric":"score (0-100)","notes":"Mean response length 2247 characters. Adjustment centered on2000characters; adjusted and unadjusted values remain distinct configurations.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":95.1,"scoreUnit":"index","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"healthbench-consensus-not-specified","evidenceKind":"lab_self_report","harnessId":"healthbench-consensus-not-specified:openai-astra-system-card:dW5hZGp1c3RlZDsgb2ZmaWNpYWwgSGVhbHRoQmVuY2ggc2NvcmluZw","id":"launch-openai-astra-system-card-524","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"unadjusted; official HealthBench scoring","family":"HealthBench Consensus","locator":"Table6 / HealthBench Consensus / gpt-5.6-terra / unadjusted","metric":"score (0-100)","notes":"Mean response length 2247 characters. Adjustment centered on2000characters; adjusted and unadjusted values remain distinct configurations.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":95.2,"scoreUnit":"index","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"healthbench-consensus-not-specified","evidenceKind":"lab_self_report","harnessId":"healthbench-consensus-not-specified:openai-astra-system-card:bGVuZ3RoLWFkanVzdGVkOyBvZmZpY2lhbCBIZWFsdGhCZW5jaCBzY29yaW5n","id":"launch-openai-astra-system-card-525","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-5.6-luna","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"length-adjusted; official HealthBench scoring","family":"HealthBench Consensus","locator":"Table6 / HealthBench Consensus / gpt-5.6-luna / length-adjusted","metric":"score (0-100)","notes":"Mean response length 1897 characters. Adjustment centered on2000characters; adjusted and unadjusted values remain distinct configurations.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":95.1,"scoreUnit":"index","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"healthbench-consensus-not-specified","evidenceKind":"lab_self_report","harnessId":"healthbench-consensus-not-specified:openai-astra-system-card:dW5hZGp1c3RlZDsgb2ZmaWNpYWwgSGVhbHRoQmVuY2ggc2NvcmluZw","id":"launch-openai-astra-system-card-526","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-5.6-luna","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"unadjusted; official HealthBench scoring","family":"HealthBench Consensus","locator":"Table6 / HealthBench Consensus / gpt-5.6-luna / unadjusted","metric":"score (0-100)","notes":"Mean response length 1897 characters. Adjustment centered on2000characters; adjusted and unadjusted values remain distinct configurations.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":95.1,"scoreUnit":"index","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"healthbench-consensus-not-specified","evidenceKind":"lab_self_report","harnessId":"healthbench-consensus-not-specified:openai-astra-system-card:bGVuZ3RoLWFkanVzdGVkOyBvZmZpY2lhbCBIZWFsdGhCZW5jaCBzY29yaW5n","id":"launch-openai-astra-system-card-527","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"length-adjusted; official HealthBench scoring","family":"HealthBench Consensus","locator":"Table6 / HealthBench Consensus / gpt-6-astra / length-adjusted","metric":"score (0-100)","notes":"Mean response length 2237 characters. Adjustment centered on2000characters; adjusted and unadjusted values remain distinct configurations.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":95.8,"scoreUnit":"index","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"healthbench-consensus-not-specified","evidenceKind":"lab_self_report","harnessId":"healthbench-consensus-not-specified:openai-astra-system-card:dW5hZGp1c3RlZDsgb2ZmaWNpYWwgSGVhbHRoQmVuY2ggc2NvcmluZw","id":"launch-openai-astra-system-card-528","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"unadjusted; official HealthBench scoring","family":"HealthBench Consensus","locator":"Table6 / HealthBench Consensus / gpt-6-astra / unadjusted","metric":"score (0-100)","notes":"Mean response length 2237 characters. Adjustment centered on2000characters; adjusted and unadjusted values remain distinct configurations.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":95.9,"scoreUnit":"index","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"internal-research-debugging-evaluation-41-research-bugs-6-alignment-auditing-tasks","evidenceKind":"lab_self_report","harnessId":"internal-research-debugging-evaluation-41-research-bugs-6-alignment-auditing-tasks:openai-astra-system-card:U3lzdGVtLWNhcmQgcmVwb3J0ZWQgY29uZmlndXJhdGlvbjsgcmVhc29uaW5nIGVmZm9ydCB1bnNwZWNpZmllZA","id":"launch-openai-astra-system-card-529","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"System-card reported configuration; reasoning effort unspecified","family":"Internal Research Debugging Evaluation","locator":"Section10.1.3.1","metric":"percent","notes":"Provider-published result; preserves source setting without claiming a controlled cross-provider comparison.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":78.05,"scoreUnit":"percent","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"sre-bench-262-binaries-19-programs","evidenceKind":"lab_self_report","harnessId":"sre-bench-262-binaries-19-programs:openai-astra-system-card:cGFzc0A0OyBmb3VyIGluZGVwZW5kZW50IHRyaWFsczsgYWxsIHNpeCBvYmplY3RpdmVzIHJlcXVpcmVkOyByZWR1Y2VkIHByb2R1Y3Rpb24gc2FmZWd1YXJkcw","id":"launch-openai-astra-system-card-530","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"pass@4; four independent trials; all six objectives required; reduced production safeguards","family":"SRE-Bench","locator":"Section10.1.2.3","metric":"percent","notes":"Provider-published result; preserves source setting without claiming a controlled cross-provider comparison.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":99.2,"scoreUnit":"percent","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"sre-bench-262-binaries-19-programs","evidenceKind":"lab_self_report","harnessId":"sre-bench-262-binaries-19-programs:openai-astra-system-card:cGFzc0A0OyBmb3VyIGluZGVwZW5kZW50IHRyaWFsczsgYWxsIHNpeCBvYmplY3RpdmVzIHJlcXVpcmVkOyByZWR1Y2VkIHByb2R1Y3Rpb24gc2FmZWd1YXJkcw","id":"launch-openai-astra-system-card-531","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"pass@4; four independent trials; all six objectives required; reduced production safeguards","family":"SRE-Bench","locator":"Section10.1.2.3","metric":"percent","notes":"Provider-published result; preserves source setting without claiming a controlled cross-provider comparison.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":68.7,"scoreUnit":"percent","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"sandbox-bench-september2026-internal","evidenceKind":"lab_self_report","harnessId":"sandbox-bench-september2026-internal:openai-astra-system-card:MjIgaXNvbGF0ZWQgQ1RGLXN0eWxlIHRhcmdldHM7IHByb3RlY3RlZC1mbGFnIHN1Y2Nlc3MgbWV0cmlj","id":"launch-openai-astra-system-card-532","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"22 isolated CTF-style targets; protected-flag success metric","family":"Sandbox Bench","locator":"Section10.1.2.4","metric":"percent","notes":"Provider-published result; preserves source setting without claiming a controlled cross-provider comparison.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":45.5,"scoreUnit":"percent","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"sandbox-bench-september2026-internal","evidenceKind":"lab_self_report","harnessId":"sandbox-bench-september2026-internal:openai-astra-system-card:MjIgaXNvbGF0ZWQgQ1RGLXN0eWxlIHRhcmdldHM7IHByb3RlY3RlZC1mbGFnIHN1Y2Nlc3MgbWV0cmlj","id":"launch-openai-astra-system-card-533","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"22 isolated CTF-style targets; protected-flag success metric","family":"Sandbox Bench","locator":"Section10.1.2.4","metric":"percent","notes":"Provider-published result; preserves source setting without claiming a controlled cross-provider comparison.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":4.5,"scoreUnit":"percent","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"no-cot-math-time-horizon-not-specified","evidenceKind":"lab_self_report","harnessId":"no-cot-math-time-horizon-not-specified:openai-astra-system-card:VUsgQUlTSTsgc2luZ2xlIGZvcndhcmQgcGFzczsgbm8gY2hhaW4gb2YgdGhvdWdodA","id":"launch-openai-astra-system-card-534","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"UK AISI; single forward pass; no chain of thought","family":"No-CoT math time horizon","locator":"Section9.3","metric":"minutes","notes":"Time-horizon estimate; possible contamination noted by evaluator. Higher means harder tasks solved, not slower inference.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":30.9,"scoreUnit":"index","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"no-cot-math-time-horizon-not-specified","evidenceKind":"lab_self_report","harnessId":"no-cot-math-time-horizon-not-specified:openai-astra-system-card:VUsgQUlTSTsgc2luZ2xlIGZvcndhcmQgcGFzczsgbm8gY2hhaW4gb2YgdGhvdWdodA","id":"launch-openai-astra-system-card-535","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"UK AISI; single forward pass; no chain of thought","family":"No-CoT math time horizon","locator":"Section9.3","metric":"minutes","notes":"Time-horizon estimate; possible contamination noted by evaluator. Higher means harder tasks solved, not slower inference.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":3.6,"scoreUnit":"index","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"protocolqa-open-ended-108-questions","evidenceKind":"lab_self_report","harnessId":"protocolqa-open-ended-108-questions:openai-astra-system-card:b2JzZXJ2ZWQgcGVyZm9ybWFuY2U","id":"launch-openai-astra-system-card-536","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"observed performance","family":"ProtocolQA Open-Ended","locator":"Section10.1.1.1 / ProtocolQA Open-Ended","metric":"percent","notes":"Provider-published result; preserves source setting without claiming a controlled cross-provider comparison.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":41.36,"scoreUnit":"percent","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"protocolqa-open-ended-108-questions","evidenceKind":"lab_self_report","harnessId":"protocolqa-open-ended-108-questions:openai-astra-system-card:cmVmdXNhbC1hZGp1c3RlZCB1cHBlciBlc3RpbWF0ZTsgcmVmdXNhbHMgY291bnRlZCBhcyBzdWNjZXNzZXM","id":"launch-openai-astra-system-card-537","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"refusal-adjusted upper estimate; refusals counted as successes","family":"ProtocolQA Open-Ended","locator":"Section10.1.1.1 / ProtocolQA Open-Ended","metric":"percent","notes":"Provider-published result; preserves source setting without claiming a controlled cross-provider comparison.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":45.37,"scoreUnit":"percent","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"tacit-knowledge-and-troubleshooting-60-questions","evidenceKind":"lab_self_report","harnessId":"tacit-knowledge-and-troubleshooting-60-questions:openai-astra-system-card:b2JzZXJ2ZWQgcGVyZm9ybWFuY2U","id":"launch-openai-astra-system-card-538","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"observed performance","family":"Tacit Knowledge and Troubleshooting","locator":"Section10.1.1.1 / Tacit Knowledge and Troubleshooting","metric":"percent","notes":"Provider-published result; preserves source setting without claiming a controlled cross-provider comparison.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":63.33,"scoreUnit":"percent","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"tacit-knowledge-and-troubleshooting-60-questions","evidenceKind":"lab_self_report","harnessId":"tacit-knowledge-and-troubleshooting-60-questions:openai-astra-system-card:cmVmdXNhbC1hZGp1c3RlZCB1cHBlciBlc3RpbWF0ZTsgcmVmdXNhbHMgY291bnRlZCBhcyBzdWNjZXNzZXM","id":"launch-openai-astra-system-card-539","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"refusal-adjusted upper estimate; refusals counted as successes","family":"Tacit Knowledge and Troubleshooting","locator":"Section10.1.1.1 / Tacit Knowledge and Troubleshooting","metric":"percent","notes":"Provider-published result; preserves source setting without claiming a controlled cross-provider comparison.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":90,"scoreUnit":"percent","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"troubleshootingbench-156-questions-52-protocols","evidenceKind":"lab_self_report","harnessId":"troubleshootingbench-156-questions-52-protocols:openai-astra-system-card:b2JzZXJ2ZWQgcGVyZm9ybWFuY2U","id":"launch-openai-astra-system-card-540","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"observed performance","family":"TroubleshootingBench","locator":"Section10.1.1.1 / TroubleshootingBench","metric":"percent","notes":"Provider-published result; preserves source setting without claiming a controlled cross-provider comparison.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":48.44,"scoreUnit":"percent","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"troubleshootingbench-156-questions-52-protocols","evidenceKind":"lab_self_report","harnessId":"troubleshootingbench-156-questions-52-protocols:openai-astra-system-card:cmVmdXNhbC1hZGp1c3RlZCB1cHBlciBlc3RpbWF0ZTsgcmVmdXNhbHMgY291bnRlZCBhcyBzdWNjZXNzZXM","id":"launch-openai-astra-system-card-541","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"refusal-adjusted upper estimate; refusals counted as successes","family":"TroubleshootingBench","locator":"Section10.1.1.1 / TroubleshootingBench","metric":"percent","notes":"Provider-published result; preserves source setting without claiming a controlled cross-provider comparison.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":63.46,"scoreUnit":"percent","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"shp2-protein-function-prediction-3-unpublished-assay-datasets","evidenceKind":"lab_self_report","harnessId":"shp2-protein-function-prediction-3-unpublished-assay-datasets:openai-astra-system-card:U3lzdGVtLWNhcmQgcmVwb3J0ZWQgY29uZmlndXJhdGlvbjsgcmVhc29uaW5nIGVmZm9ydCB1bnNwZWNpZmllZA","id":"launch-openai-astra-system-card-542","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"System-card reported configuration; reasoning effort unspecified","family":"SHP2 Protein Function Prediction","locator":"Section10.1.1.2.2","metric":"mean R-squared","notes":"Production-named model result; separate helpful-only checkpoint omitted because no exact catalog identity.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":0.4,"scoreUnit":"index","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"shp2-protein-function-prediction-3-unpublished-assay-datasets","evidenceKind":"lab_self_report","harnessId":"shp2-protein-function-prediction-3-unpublished-assay-datasets:openai-astra-system-card:U3lzdGVtLWNhcmQgcmVwb3J0ZWQgY29uZmlndXJhdGlvbjsgcmVhc29uaW5nIGVmZm9ydCB1bnNwZWNpZmllZA","id":"launch-openai-astra-system-card-543","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"System-card reported configuration; reasoning effort unspecified","family":"SHP2 Protein Function Prediction","locator":"Section10.1.1.2.2","metric":"mean R-squared","notes":"Production-named model result; separate helpful-only checkpoint omitted because no exact catalog identity.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":0.3,"scoreUnit":"index","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"coronavirus-ace2-cell-entry-screen-not-specified","evidenceKind":"lab_self_report","harnessId":"coronavirus-ace2-cell-entry-screen-not-specified:openai-astra-system-card:U3lzdGVtLWNhcmQgcmVwb3J0ZWQgY29uZmlndXJhdGlvbjsgcmVhc29uaW5nIGVmZm9ydCB1bnNwZWNpZmllZA","id":"launch-openai-astra-system-card-544","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"System-card reported configuration; reasoning effort unspecified","family":"Coronavirus-ACE2 Cell-Entry Screen","locator":"Section10.1.1.2.3","metric":"composite score","notes":"Observed named-model performance; helpful-only checkpoint omitted as distinct noncatalog identity.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":0.42,"scoreUnit":"index","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"coronavirus-ace2-cell-entry-screen-not-specified","evidenceKind":"lab_self_report","harnessId":"coronavirus-ace2-cell-entry-screen-not-specified:openai-astra-system-card:U3lzdGVtLWNhcmQgcmVwb3J0ZWQgY29uZmlndXJhdGlvbjsgcmVhc29uaW5nIGVmZm9ydCB1bnNwZWNpZmllZA","id":"launch-openai-astra-system-card-545","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"System-card reported configuration; reasoning effort unspecified","family":"Coronavirus-ACE2 Cell-Entry Screen","locator":"Section10.1.1.2.3","metric":"composite score","notes":"Observed named-model performance; helpful-only checkpoint omitted as distinct noncatalog identity.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":0.43,"scoreUnit":"index","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"phage-plasmid-co-evolution-not-specified","evidenceKind":"lab_self_report","harnessId":"phage-plasmid-co-evolution-not-specified:openai-astra-system-card:U3lzdGVtLWNhcmQgcmVwb3J0ZWQgY29uZmlndXJhdGlvbjsgcmVhc29uaW5nIGVmZm9ydCB1bnNwZWNpZmllZA","id":"launch-openai-astra-system-card-546","ingestRunId":"launch-openai-astra-system-card","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"System-card reported configuration; reasoning effort unspecified","family":"Phage-plasmid Co-evolution","locator":"Section10.1.1.2.4","metric":"negative log-likelihood","notes":"Lower is better. Helpful-only checkpoint omitted as distinct noncatalog identity.","sourceId":"openai-astra-system-card","sourceTitle":"GPT-6 Astra System Card"},"score":13.13,"scoreUnit":"index","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra"},{"benchmarkId":"automationbench-1-0-6-percent","evidenceKind":"lab_self_report","harnessId":"automationbench-1-0-6-percent:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBsb3cuIEF1dG9tYXRpb25CZW5jaCAxLjAuNi4gRW5kLXRvLWVuZCB3b3JrZmxvd3MgYWNyb3NzIDQ3IHRvb2xzIGluIHNhbGVzLCBtYXJrZXRpbmcsIG9wZXJhdGlvbnMsIHN1cHBvcnQsIGZpbmFuY2UsIGFuZCBIUi4gVGhlIHBhZ2Ugc2F5cyB0aGUgQ2xhdWRlIEZhYmxlIDUuMSBjb3N0IHBvaW50IG9taXRzIE9wdXMgNSBmYWxsYmFja3MuIE9wZW5BSS1wdWJsaXNoZWQgY2hhcnQgb24gdGhlIEludHJvZHVjaW5nIEdQVC02IFNvbCBhbmQgTHVuYSBwYWdlLg","id":"launch-openai-gpt6-sol-luna-launch-2306","ingestRunId":"launch-openai-gpt6-sol-luna-launch","modelId":"gpt-6-luna","n":null,"observedOn":"2026-09-22","rawPayloadHash":null,"report":{"category":"agentic","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Reported reasoning effort low. AutomationBench 1.0.6. End-to-end workflows across 47 tools in sales, marketing, operations, support, finance, and HR. The page says the Claude Fable 5.1 cost point omits Opus 5 fallbacks. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page.","family":"AutomationBench","locator":"Chart: AutomationBench / GPT-6 Luna / low","metric":"percent","notes":"Provider-published lab self-report. Vega-Lite chart data value 0.012, stored as 1.2 percent. Not an independent board.","sourceId":"openai-gpt6-sol-luna-launch","sourceTitle":"Introducing GPT-6 Sol and Luna"},"score":1.2,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-sol-and-luna/"},{"benchmarkId":"automationbench-1-0-6-percent","evidenceKind":"lab_self_report","harnessId":"automationbench-1-0-6-percent:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtZWRpdW0uIEF1dG9tYXRpb25CZW5jaCAxLjAuNi4gRW5kLXRvLWVuZCB3b3JrZmxvd3MgYWNyb3NzIDQ3IHRvb2xzIGluIHNhbGVzLCBtYXJrZXRpbmcsIG9wZXJhdGlvbnMsIHN1cHBvcnQsIGZpbmFuY2UsIGFuZCBIUi4gVGhlIHBhZ2Ugc2F5cyB0aGUgQ2xhdWRlIEZhYmxlIDUuMSBjb3N0IHBvaW50IG9taXRzIE9wdXMgNSBmYWxsYmFja3MuIE9wZW5BSS1wdWJsaXNoZWQgY2hhcnQgb24gdGhlIEludHJvZHVjaW5nIEdQVC02IFNvbCBhbmQgTHVuYSBwYWdlLg","id":"launch-openai-gpt6-sol-luna-launch-2307","ingestRunId":"launch-openai-gpt6-sol-luna-launch","modelId":"gpt-6-luna","n":null,"observedOn":"2026-09-22","rawPayloadHash":null,"report":{"category":"agentic","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Reported reasoning effort medium. AutomationBench 1.0.6. End-to-end workflows across 47 tools in sales, marketing, operations, support, finance, and HR. The page says the Claude Fable 5.1 cost point omits Opus 5 fallbacks. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page.","family":"AutomationBench","locator":"Chart: AutomationBench / GPT-6 Luna / medium","metric":"percent","notes":"Provider-published lab self-report. Vega-Lite chart data value 0.094, stored as 9.4 percent. Not an independent board.","sourceId":"openai-gpt6-sol-luna-launch","sourceTitle":"Introducing GPT-6 Sol and Luna"},"score":9.4,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-sol-and-luna/"},{"benchmarkId":"automationbench-1-0-6-percent","evidenceKind":"lab_self_report","harnessId":"automationbench-1-0-6-percent:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBoaWdoLiBBdXRvbWF0aW9uQmVuY2ggMS4wLjYuIEVuZC10by1lbmQgd29ya2Zsb3dzIGFjcm9zcyA0NyB0b29scyBpbiBzYWxlcywgbWFya2V0aW5nLCBvcGVyYXRpb25zLCBzdXBwb3J0LCBmaW5hbmNlLCBhbmQgSFIuIFRoZSBwYWdlIHNheXMgdGhlIENsYXVkZSBGYWJsZSA1LjEgY29zdCBwb2ludCBvbWl0cyBPcHVzIDUgZmFsbGJhY2tzLiBPcGVuQUktcHVibGlzaGVkIGNoYXJ0IG9uIHRoZSBJbnRyb2R1Y2luZyBHUFQtNiBTb2wgYW5kIEx1bmEgcGFnZS4","id":"launch-openai-gpt6-sol-luna-launch-2308","ingestRunId":"launch-openai-gpt6-sol-luna-launch","modelId":"gpt-6-luna","n":null,"observedOn":"2026-09-22","rawPayloadHash":null,"report":{"category":"agentic","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Reported reasoning effort high. AutomationBench 1.0.6. End-to-end workflows across 47 tools in sales, marketing, operations, support, finance, and HR. The page says the Claude Fable 5.1 cost point omits Opus 5 fallbacks. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page.","family":"AutomationBench","locator":"Chart: AutomationBench / GPT-6 Luna / high","metric":"percent","notes":"Provider-published lab self-report. Vega-Lite chart data value 0.145, stored as 14.5 percent. Not an independent board.","sourceId":"openai-gpt6-sol-luna-launch","sourceTitle":"Introducing GPT-6 Sol and Luna"},"score":14.5,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-sol-and-luna/"},{"benchmarkId":"automationbench-1-0-6-percent","evidenceKind":"lab_self_report","harnessId":"automationbench-1-0-6-percent:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCB4aGlnaC4gQXV0b21hdGlvbkJlbmNoIDEuMC42LiBFbmQtdG8tZW5kIHdvcmtmbG93cyBhY3Jvc3MgNDcgdG9vbHMgaW4gc2FsZXMsIG1hcmtldGluZywgb3BlcmF0aW9ucywgc3VwcG9ydCwgZmluYW5jZSwgYW5kIEhSLiBUaGUgcGFnZSBzYXlzIHRoZSBDbGF1ZGUgRmFibGUgNS4xIGNvc3QgcG9pbnQgb21pdHMgT3B1cyA1IGZhbGxiYWNrcy4gT3BlbkFJLXB1Ymxpc2hlZCBjaGFydCBvbiB0aGUgSW50cm9kdWNpbmcgR1BULTYgU29sIGFuZCBMdW5hIHBhZ2Uu","id":"launch-openai-gpt6-sol-luna-launch-2309","ingestRunId":"launch-openai-gpt6-sol-luna-launch","modelId":"gpt-6-luna","n":null,"observedOn":"2026-09-22","rawPayloadHash":null,"report":{"category":"agentic","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Reported reasoning effort xhigh. AutomationBench 1.0.6. End-to-end workflows across 47 tools in sales, marketing, operations, support, finance, and HR. The page says the Claude Fable 5.1 cost point omits Opus 5 fallbacks. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page.","family":"AutomationBench","locator":"Chart: AutomationBench / GPT-6 Luna / xhigh","metric":"percent","notes":"Provider-published lab self-report. Vega-Lite chart data value 0.126, stored as 12.6 percent. Not an independent board.","sourceId":"openai-gpt6-sol-luna-launch","sourceTitle":"Introducing GPT-6 Sol and Luna"},"score":12.6,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-sol-and-luna/"},{"benchmarkId":"automationbench-1-0-6-percent","evidenceKind":"lab_self_report","harnessId":"automationbench-1-0-6-percent:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIEF1dG9tYXRpb25CZW5jaCAxLjAuNi4gRW5kLXRvLWVuZCB3b3JrZmxvd3MgYWNyb3NzIDQ3IHRvb2xzIGluIHNhbGVzLCBtYXJrZXRpbmcsIG9wZXJhdGlvbnMsIHN1cHBvcnQsIGZpbmFuY2UsIGFuZCBIUi4gVGhlIHBhZ2Ugc2F5cyB0aGUgQ2xhdWRlIEZhYmxlIDUuMSBjb3N0IHBvaW50IG9taXRzIE9wdXMgNSBmYWxsYmFja3MuIE9wZW5BSS1wdWJsaXNoZWQgY2hhcnQgb24gdGhlIEludHJvZHVjaW5nIEdQVC02IFNvbCBhbmQgTHVuYSBwYWdlLg","id":"launch-openai-gpt6-sol-luna-launch-2310","ingestRunId":"launch-openai-gpt6-sol-luna-launch","modelId":"gpt-6-luna","n":null,"observedOn":"2026-09-22","rawPayloadHash":null,"report":{"category":"agentic","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Reported reasoning effort max. AutomationBench 1.0.6. End-to-end workflows across 47 tools in sales, marketing, operations, support, finance, and HR. The page says the Claude Fable 5.1 cost point omits Opus 5 fallbacks. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page.","family":"AutomationBench","locator":"Chart: AutomationBench / GPT-6 Luna / max","metric":"percent","notes":"Provider-published lab self-report. Vega-Lite chart data value 0.207, stored as 20.7 percent. Not an independent board.","sourceId":"openai-gpt6-sol-luna-launch","sourceTitle":"Introducing GPT-6 Sol and Luna"},"score":20.7,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-sol-and-luna/"},{"benchmarkId":"automationbench-1-0-6-percent","evidenceKind":"lab_self_report","harnessId":"automationbench-1-0-6-percent:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBsb3cuIEF1dG9tYXRpb25CZW5jaCAxLjAuNi4gRW5kLXRvLWVuZCB3b3JrZmxvd3MgYWNyb3NzIDQ3IHRvb2xzIGluIHNhbGVzLCBtYXJrZXRpbmcsIG9wZXJhdGlvbnMsIHN1cHBvcnQsIGZpbmFuY2UsIGFuZCBIUi4gVGhlIHBhZ2Ugc2F5cyB0aGUgQ2xhdWRlIEZhYmxlIDUuMSBjb3N0IHBvaW50IG9taXRzIE9wdXMgNSBmYWxsYmFja3MuIE9wZW5BSS1wdWJsaXNoZWQgY2hhcnQgb24gdGhlIEludHJvZHVjaW5nIEdQVC02IFNvbCBhbmQgTHVuYSBwYWdlLg","id":"launch-openai-gpt6-sol-luna-launch-2311","ingestRunId":"launch-openai-gpt6-sol-luna-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-22","rawPayloadHash":null,"report":{"category":"agentic","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Reported reasoning effort low. AutomationBench 1.0.6. End-to-end workflows across 47 tools in sales, marketing, operations, support, finance, and HR. The page says the Claude Fable 5.1 cost point omits Opus 5 fallbacks. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page.","family":"AutomationBench","locator":"Chart: AutomationBench / GPT-6 Sol / low","metric":"percent","notes":"Provider-published lab self-report. Vega-Lite chart data value 0.2116, stored as 21.16 percent. Not an independent board.","sourceId":"openai-gpt6-sol-luna-launch","sourceTitle":"Introducing GPT-6 Sol and Luna"},"score":21.16,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-sol-and-luna/"},{"benchmarkId":"automationbench-1-0-6-percent","evidenceKind":"lab_self_report","harnessId":"automationbench-1-0-6-percent:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtZWRpdW0uIEF1dG9tYXRpb25CZW5jaCAxLjAuNi4gRW5kLXRvLWVuZCB3b3JrZmxvd3MgYWNyb3NzIDQ3IHRvb2xzIGluIHNhbGVzLCBtYXJrZXRpbmcsIG9wZXJhdGlvbnMsIHN1cHBvcnQsIGZpbmFuY2UsIGFuZCBIUi4gVGhlIHBhZ2Ugc2F5cyB0aGUgQ2xhdWRlIEZhYmxlIDUuMSBjb3N0IHBvaW50IG9taXRzIE9wdXMgNSBmYWxsYmFja3MuIE9wZW5BSS1wdWJsaXNoZWQgY2hhcnQgb24gdGhlIEludHJvZHVjaW5nIEdQVC02IFNvbCBhbmQgTHVuYSBwYWdlLg","id":"launch-openai-gpt6-sol-luna-launch-2312","ingestRunId":"launch-openai-gpt6-sol-luna-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-22","rawPayloadHash":null,"report":{"category":"agentic","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Reported reasoning effort medium. AutomationBench 1.0.6. End-to-end workflows across 47 tools in sales, marketing, operations, support, finance, and HR. The page says the Claude Fable 5.1 cost point omits Opus 5 fallbacks. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page.","family":"AutomationBench","locator":"Chart: AutomationBench / GPT-6 Sol / medium","metric":"percent","notes":"Provider-published lab self-report. Vega-Lite chart data value 0.2694, stored as 26.94 percent. Not an independent board.","sourceId":"openai-gpt6-sol-luna-launch","sourceTitle":"Introducing GPT-6 Sol and Luna"},"score":26.94,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-sol-and-luna/"},{"benchmarkId":"automationbench-1-0-6-percent","evidenceKind":"lab_self_report","harnessId":"automationbench-1-0-6-percent:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBoaWdoLiBBdXRvbWF0aW9uQmVuY2ggMS4wLjYuIEVuZC10by1lbmQgd29ya2Zsb3dzIGFjcm9zcyA0NyB0b29scyBpbiBzYWxlcywgbWFya2V0aW5nLCBvcGVyYXRpb25zLCBzdXBwb3J0LCBmaW5hbmNlLCBhbmQgSFIuIFRoZSBwYWdlIHNheXMgdGhlIENsYXVkZSBGYWJsZSA1LjEgY29zdCBwb2ludCBvbWl0cyBPcHVzIDUgZmFsbGJhY2tzLiBPcGVuQUktcHVibGlzaGVkIGNoYXJ0IG9uIHRoZSBJbnRyb2R1Y2luZyBHUFQtNiBTb2wgYW5kIEx1bmEgcGFnZS4","id":"launch-openai-gpt6-sol-luna-launch-2313","ingestRunId":"launch-openai-gpt6-sol-luna-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-22","rawPayloadHash":null,"report":{"category":"agentic","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Reported reasoning effort high. AutomationBench 1.0.6. End-to-end workflows across 47 tools in sales, marketing, operations, support, finance, and HR. The page says the Claude Fable 5.1 cost point omits Opus 5 fallbacks. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page.","family":"AutomationBench","locator":"Chart: AutomationBench / GPT-6 Sol / high","metric":"percent","notes":"Provider-published lab self-report. Vega-Lite chart data value 0.312, stored as 31.2 percent. Not an independent board.","sourceId":"openai-gpt6-sol-luna-launch","sourceTitle":"Introducing GPT-6 Sol and Luna"},"score":31.2,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-sol-and-luna/"},{"benchmarkId":"automationbench-1-0-6-percent","evidenceKind":"lab_self_report","harnessId":"automationbench-1-0-6-percent:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCB4aGlnaC4gQXV0b21hdGlvbkJlbmNoIDEuMC42LiBFbmQtdG8tZW5kIHdvcmtmbG93cyBhY3Jvc3MgNDcgdG9vbHMgaW4gc2FsZXMsIG1hcmtldGluZywgb3BlcmF0aW9ucywgc3VwcG9ydCwgZmluYW5jZSwgYW5kIEhSLiBUaGUgcGFnZSBzYXlzIHRoZSBDbGF1ZGUgRmFibGUgNS4xIGNvc3QgcG9pbnQgb21pdHMgT3B1cyA1IGZhbGxiYWNrcy4gT3BlbkFJLXB1Ymxpc2hlZCBjaGFydCBvbiB0aGUgSW50cm9kdWNpbmcgR1BULTYgU29sIGFuZCBMdW5hIHBhZ2Uu","id":"launch-openai-gpt6-sol-luna-launch-2314","ingestRunId":"launch-openai-gpt6-sol-luna-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-22","rawPayloadHash":null,"report":{"category":"agentic","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Reported reasoning effort xhigh. AutomationBench 1.0.6. End-to-end workflows across 47 tools in sales, marketing, operations, support, finance, and HR. The page says the Claude Fable 5.1 cost point omits Opus 5 fallbacks. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page.","family":"AutomationBench","locator":"Chart: AutomationBench / GPT-6 Sol / xhigh","metric":"percent","notes":"Provider-published lab self-report. Vega-Lite chart data value 0.3318, stored as 33.18 percent. Not an independent board.","sourceId":"openai-gpt6-sol-luna-launch","sourceTitle":"Introducing GPT-6 Sol and Luna"},"score":33.18,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-sol-and-luna/"},{"benchmarkId":"automationbench-1-0-6-percent","evidenceKind":"lab_self_report","harnessId":"automationbench-1-0-6-percent:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIEF1dG9tYXRpb25CZW5jaCAxLjAuNi4gRW5kLXRvLWVuZCB3b3JrZmxvd3MgYWNyb3NzIDQ3IHRvb2xzIGluIHNhbGVzLCBtYXJrZXRpbmcsIG9wZXJhdGlvbnMsIHN1cHBvcnQsIGZpbmFuY2UsIGFuZCBIUi4gVGhlIHBhZ2Ugc2F5cyB0aGUgQ2xhdWRlIEZhYmxlIDUuMSBjb3N0IHBvaW50IG9taXRzIE9wdXMgNSBmYWxsYmFja3MuIE9wZW5BSS1wdWJsaXNoZWQgY2hhcnQgb24gdGhlIEludHJvZHVjaW5nIEdQVC02IFNvbCBhbmQgTHVuYSBwYWdlLg","id":"launch-openai-gpt6-sol-luna-launch-2315","ingestRunId":"launch-openai-gpt6-sol-luna-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-22","rawPayloadHash":null,"report":{"category":"agentic","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Reported reasoning effort max. AutomationBench 1.0.6. End-to-end workflows across 47 tools in sales, marketing, operations, support, finance, and HR. The page says the Claude Fable 5.1 cost point omits Opus 5 fallbacks. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page.","family":"AutomationBench","locator":"Chart: AutomationBench / GPT-6 Sol / max","metric":"percent","notes":"Provider-published lab self-report. Vega-Lite chart data value 0.3196, stored as 31.96 percent. Not an independent board.","sourceId":"openai-gpt6-sol-luna-launch","sourceTitle":"Introducing GPT-6 Sol and Luna"},"score":31.96,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-sol-and-luna/"},{"benchmarkId":"agents-last-exam-1","evidenceKind":"lab_self_report","harnessId":"agents-last-exam-1:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBsb3cuIEFnZW50cycgTGFzdCBFeGFtIFYxLiBMb25nLWhvcml6b24gcHJvZmVzc2lvbmFsIHRhc2tzIGFjcm9zcyA1NSBzdWItaW5kdXN0cmllcy4gT3BlbkFJLXB1Ymxpc2hlZCBjaGFydCBvbiB0aGUgSW50cm9kdWNpbmcgR1BULTYgU29sIGFuZCBMdW5hIHBhZ2Uu","id":"launch-openai-gpt6-sol-luna-launch-2316","ingestRunId":"launch-openai-gpt6-sol-luna-launch","modelId":"gpt-6-luna","n":null,"observedOn":"2026-09-22","rawPayloadHash":null,"report":{"category":"agentic","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Reported reasoning effort low. Agents' Last Exam V1. Long-horizon professional tasks across 55 sub-industries. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page.","family":"Agents' Last Exam","locator":"Chart: Agents' Last Exam / GPT-6 Luna / low","metric":"percent","notes":"Provider-published lab self-report. Vega-Lite chart data value 0.3632, stored as 36.32 percent. Not an independent board.","sourceId":"openai-gpt6-sol-luna-launch","sourceTitle":"Introducing GPT-6 Sol and Luna"},"score":36.32,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-sol-and-luna/"},{"benchmarkId":"agents-last-exam-1","evidenceKind":"lab_self_report","harnessId":"agents-last-exam-1:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtZWRpdW0uIEFnZW50cycgTGFzdCBFeGFtIFYxLiBMb25nLWhvcml6b24gcHJvZmVzc2lvbmFsIHRhc2tzIGFjcm9zcyA1NSBzdWItaW5kdXN0cmllcy4gT3BlbkFJLXB1Ymxpc2hlZCBjaGFydCBvbiB0aGUgSW50cm9kdWNpbmcgR1BULTYgU29sIGFuZCBMdW5hIHBhZ2Uu","id":"launch-openai-gpt6-sol-luna-launch-2317","ingestRunId":"launch-openai-gpt6-sol-luna-launch","modelId":"gpt-6-luna","n":null,"observedOn":"2026-09-22","rawPayloadHash":null,"report":{"category":"agentic","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Reported reasoning effort medium. Agents' Last Exam V1. Long-horizon professional tasks across 55 sub-industries. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page.","family":"Agents' Last Exam","locator":"Chart: Agents' Last Exam / GPT-6 Luna / medium","metric":"percent","notes":"Provider-published lab self-report. Vega-Lite chart data value 0.4684, stored as 46.84 percent. Not an independent board.","sourceId":"openai-gpt6-sol-luna-launch","sourceTitle":"Introducing GPT-6 Sol and Luna"},"score":46.84,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-sol-and-luna/"},{"benchmarkId":"agents-last-exam-1","evidenceKind":"lab_self_report","harnessId":"agents-last-exam-1:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBoaWdoLiBBZ2VudHMnIExhc3QgRXhhbSBWMS4gTG9uZy1ob3Jpem9uIHByb2Zlc3Npb25hbCB0YXNrcyBhY3Jvc3MgNTUgc3ViLWluZHVzdHJpZXMuIE9wZW5BSS1wdWJsaXNoZWQgY2hhcnQgb24gdGhlIEludHJvZHVjaW5nIEdQVC02IFNvbCBhbmQgTHVuYSBwYWdlLg","id":"launch-openai-gpt6-sol-luna-launch-2318","ingestRunId":"launch-openai-gpt6-sol-luna-launch","modelId":"gpt-6-luna","n":null,"observedOn":"2026-09-22","rawPayloadHash":null,"report":{"category":"agentic","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Reported reasoning effort high. Agents' Last Exam V1. Long-horizon professional tasks across 55 sub-industries. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page.","family":"Agents' Last Exam","locator":"Chart: Agents' Last Exam / GPT-6 Luna / high","metric":"percent","notes":"Provider-published lab self-report. Vega-Lite chart data value 0.436, stored as 43.6 percent. Not an independent board.","sourceId":"openai-gpt6-sol-luna-launch","sourceTitle":"Introducing GPT-6 Sol and Luna"},"score":43.6,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-sol-and-luna/"},{"benchmarkId":"agents-last-exam-1","evidenceKind":"lab_self_report","harnessId":"agents-last-exam-1:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCB4aGlnaC4gQWdlbnRzJyBMYXN0IEV4YW0gVjEuIExvbmctaG9yaXpvbiBwcm9mZXNzaW9uYWwgdGFza3MgYWNyb3NzIDU1IHN1Yi1pbmR1c3RyaWVzLiBPcGVuQUktcHVibGlzaGVkIGNoYXJ0IG9uIHRoZSBJbnRyb2R1Y2luZyBHUFQtNiBTb2wgYW5kIEx1bmEgcGFnZS4","id":"launch-openai-gpt6-sol-luna-launch-2319","ingestRunId":"launch-openai-gpt6-sol-luna-launch","modelId":"gpt-6-luna","n":null,"observedOn":"2026-09-22","rawPayloadHash":null,"report":{"category":"agentic","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Reported reasoning effort xhigh. Agents' Last Exam V1. Long-horizon professional tasks across 55 sub-industries. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page.","family":"Agents' Last Exam","locator":"Chart: Agents' Last Exam / GPT-6 Luna / xhigh","metric":"percent","notes":"Provider-published lab self-report. Vega-Lite chart data value 0.4789, stored as 47.89 percent. Not an independent board.","sourceId":"openai-gpt6-sol-luna-launch","sourceTitle":"Introducing GPT-6 Sol and Luna"},"score":47.89,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-sol-and-luna/"},{"benchmarkId":"agents-last-exam-1","evidenceKind":"lab_self_report","harnessId":"agents-last-exam-1:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIEFnZW50cycgTGFzdCBFeGFtIFYxLiBMb25nLWhvcml6b24gcHJvZmVzc2lvbmFsIHRhc2tzIGFjcm9zcyA1NSBzdWItaW5kdXN0cmllcy4gT3BlbkFJLXB1Ymxpc2hlZCBjaGFydCBvbiB0aGUgSW50cm9kdWNpbmcgR1BULTYgU29sIGFuZCBMdW5hIHBhZ2Uu","id":"launch-openai-gpt6-sol-luna-launch-2320","ingestRunId":"launch-openai-gpt6-sol-luna-launch","modelId":"gpt-6-luna","n":null,"observedOn":"2026-09-22","rawPayloadHash":null,"report":{"category":"agentic","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Reported reasoning effort max. Agents' Last Exam V1. Long-horizon professional tasks across 55 sub-industries. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page.","family":"Agents' Last Exam","locator":"Chart: Agents' Last Exam / GPT-6 Luna / max","metric":"percent","notes":"Provider-published lab self-report. Vega-Lite chart data value 0.5089, stored as 50.89 percent. Not an independent board.","sourceId":"openai-gpt6-sol-luna-launch","sourceTitle":"Introducing GPT-6 Sol and Luna"},"score":50.89,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-sol-and-luna/"},{"benchmarkId":"agents-last-exam-1","evidenceKind":"lab_self_report","harnessId":"agents-last-exam-1:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBsb3cuIEFnZW50cycgTGFzdCBFeGFtIFYxLiBMb25nLWhvcml6b24gcHJvZmVzc2lvbmFsIHRhc2tzIGFjcm9zcyA1NSBzdWItaW5kdXN0cmllcy4gT3BlbkFJLXB1Ymxpc2hlZCBjaGFydCBvbiB0aGUgSW50cm9kdWNpbmcgR1BULTYgU29sIGFuZCBMdW5hIHBhZ2Uu","id":"launch-openai-gpt6-sol-luna-launch-2321","ingestRunId":"launch-openai-gpt6-sol-luna-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-22","rawPayloadHash":null,"report":{"category":"agentic","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Reported reasoning effort low. Agents' Last Exam V1. Long-horizon professional tasks across 55 sub-industries. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page.","family":"Agents' Last Exam","locator":"Chart: Agents' Last Exam / GPT-6 Sol / low","metric":"percent","notes":"Provider-published lab self-report. Vega-Lite chart data value 0.4868, stored as 48.68 percent. Not an independent board.","sourceId":"openai-gpt6-sol-luna-launch","sourceTitle":"Introducing GPT-6 Sol and Luna"},"score":48.68,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-sol-and-luna/"},{"benchmarkId":"agents-last-exam-1","evidenceKind":"lab_self_report","harnessId":"agents-last-exam-1:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtZWRpdW0uIEFnZW50cycgTGFzdCBFeGFtIFYxLiBMb25nLWhvcml6b24gcHJvZmVzc2lvbmFsIHRhc2tzIGFjcm9zcyA1NSBzdWItaW5kdXN0cmllcy4gT3BlbkFJLXB1Ymxpc2hlZCBjaGFydCBvbiB0aGUgSW50cm9kdWNpbmcgR1BULTYgU29sIGFuZCBMdW5hIHBhZ2Uu","id":"launch-openai-gpt6-sol-luna-launch-2322","ingestRunId":"launch-openai-gpt6-sol-luna-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-22","rawPayloadHash":null,"report":{"category":"agentic","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Reported reasoning effort medium. Agents' Last Exam V1. Long-horizon professional tasks across 55 sub-industries. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page.","family":"Agents' Last Exam","locator":"Chart: Agents' Last Exam / GPT-6 Sol / medium","metric":"percent","notes":"Provider-published lab self-report. Vega-Lite chart data value 0.5306, stored as 53.06 percent. Not an independent board.","sourceId":"openai-gpt6-sol-luna-launch","sourceTitle":"Introducing GPT-6 Sol and Luna"},"score":53.06,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-sol-and-luna/"},{"benchmarkId":"agents-last-exam-1","evidenceKind":"lab_self_report","harnessId":"agents-last-exam-1:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBoaWdoLiBBZ2VudHMnIExhc3QgRXhhbSBWMS4gTG9uZy1ob3Jpem9uIHByb2Zlc3Npb25hbCB0YXNrcyBhY3Jvc3MgNTUgc3ViLWluZHVzdHJpZXMuIE9wZW5BSS1wdWJsaXNoZWQgY2hhcnQgb24gdGhlIEludHJvZHVjaW5nIEdQVC02IFNvbCBhbmQgTHVuYSBwYWdlLg","id":"launch-openai-gpt6-sol-luna-launch-2323","ingestRunId":"launch-openai-gpt6-sol-luna-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-22","rawPayloadHash":null,"report":{"category":"agentic","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Reported reasoning effort high. Agents' Last Exam V1. Long-horizon professional tasks across 55 sub-industries. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page.","family":"Agents' Last Exam","locator":"Chart: Agents' Last Exam / GPT-6 Sol / high","metric":"percent","notes":"Provider-published lab self-report. Vega-Lite chart data value 0.5258, stored as 52.58 percent. Not an independent board.","sourceId":"openai-gpt6-sol-luna-launch","sourceTitle":"Introducing GPT-6 Sol and Luna"},"score":52.58,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-sol-and-luna/"},{"benchmarkId":"agents-last-exam-1","evidenceKind":"lab_self_report","harnessId":"agents-last-exam-1:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCB4aGlnaC4gQWdlbnRzJyBMYXN0IEV4YW0gVjEuIExvbmctaG9yaXpvbiBwcm9mZXNzaW9uYWwgdGFza3MgYWNyb3NzIDU1IHN1Yi1pbmR1c3RyaWVzLiBPcGVuQUktcHVibGlzaGVkIGNoYXJ0IG9uIHRoZSBJbnRyb2R1Y2luZyBHUFQtNiBTb2wgYW5kIEx1bmEgcGFnZS4","id":"launch-openai-gpt6-sol-luna-launch-2324","ingestRunId":"launch-openai-gpt6-sol-luna-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-22","rawPayloadHash":null,"report":{"category":"agentic","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Reported reasoning effort xhigh. Agents' Last Exam V1. Long-horizon professional tasks across 55 sub-industries. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page.","family":"Agents' Last Exam","locator":"Chart: Agents' Last Exam / GPT-6 Sol / xhigh","metric":"percent","notes":"Provider-published lab self-report. Vega-Lite chart data value 0.5539, stored as 55.39 percent. Not an independent board.","sourceId":"openai-gpt6-sol-luna-launch","sourceTitle":"Introducing GPT-6 Sol and Luna"},"score":55.39,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-sol-and-luna/"},{"benchmarkId":"agents-last-exam-1","evidenceKind":"lab_self_report","harnessId":"agents-last-exam-1:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIEFnZW50cycgTGFzdCBFeGFtIFYxLiBMb25nLWhvcml6b24gcHJvZmVzc2lvbmFsIHRhc2tzIGFjcm9zcyA1NSBzdWItaW5kdXN0cmllcy4gT3BlbkFJLXB1Ymxpc2hlZCBjaGFydCBvbiB0aGUgSW50cm9kdWNpbmcgR1BULTYgU29sIGFuZCBMdW5hIHBhZ2Uu","id":"launch-openai-gpt6-sol-luna-launch-2325","ingestRunId":"launch-openai-gpt6-sol-luna-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-22","rawPayloadHash":null,"report":{"category":"agentic","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Reported reasoning effort max. Agents' Last Exam V1. Long-horizon professional tasks across 55 sub-industries. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page.","family":"Agents' Last Exam","locator":"Chart: Agents' Last Exam / GPT-6 Sol / max","metric":"percent","notes":"Provider-published lab self-report. Vega-Lite chart data value 0.5636, stored as 56.36 percent. Not an independent board. Article prose rounds this max point to 56.4%.","sourceId":"openai-gpt6-sol-luna-launch","sourceTitle":"Introducing GPT-6 Sol and Luna"},"score":56.36,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-sol-and-luna/"},{"benchmarkId":"frontiercode-1-1-main-score-1-1-main","evidenceKind":"lab_self_report","harnessId":"frontiercode-1-1-main-score-1-1-main:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBsb3cuIEZyb250aWVyQ29kZSAxLjEgTWFpbiwgYXMgaWRlbnRpZmllZCBpbiB0aGUgc3Vycm91bmRpbmcgYXJ0aWNsZS4gVGhlIGNoYXJ0IHRpdGxlIGlzIEZyb250aWVyQ29kZS4gR3JhZGVkIG9uIGNvcnJlY3RuZXNzIGFuZCBtZXJnZWFiaWxpdHkuIE9wZW5BSS1wdWJsaXNoZWQgY2hhcnQgb24gdGhlIEludHJvZHVjaW5nIEdQVC02IFNvbCBhbmQgTHVuYSBwYWdlLg","id":"launch-openai-gpt6-sol-luna-launch-2326","ingestRunId":"launch-openai-gpt6-sol-luna-launch","modelId":"gpt-6-luna","n":null,"observedOn":"2026-09-22","rawPayloadHash":null,"report":{"category":"coding","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Reported reasoning effort low. FrontierCode 1.1 Main, as identified in the surrounding article. The chart title is FrontierCode. Graded on correctness and mergeability. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page.","family":"FrontierCode 1.1 Main (score)","locator":"Chart: FrontierCode / GPT-6 Luna / low","metric":"percent","notes":"Provider-published lab self-report. Vega-Lite chart data value 0.2566, stored as 25.66 percent. Not an independent board.","sourceId":"openai-gpt6-sol-luna-launch","sourceTitle":"Introducing GPT-6 Sol and Luna"},"score":25.66,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-sol-and-luna/"},{"benchmarkId":"frontiercode-1-1-main-score-1-1-main","evidenceKind":"lab_self_report","harnessId":"frontiercode-1-1-main-score-1-1-main:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtZWRpdW0uIEZyb250aWVyQ29kZSAxLjEgTWFpbiwgYXMgaWRlbnRpZmllZCBpbiB0aGUgc3Vycm91bmRpbmcgYXJ0aWNsZS4gVGhlIGNoYXJ0IHRpdGxlIGlzIEZyb250aWVyQ29kZS4gR3JhZGVkIG9uIGNvcnJlY3RuZXNzIGFuZCBtZXJnZWFiaWxpdHkuIE9wZW5BSS1wdWJsaXNoZWQgY2hhcnQgb24gdGhlIEludHJvZHVjaW5nIEdQVC02IFNvbCBhbmQgTHVuYSBwYWdlLg","id":"launch-openai-gpt6-sol-luna-launch-2327","ingestRunId":"launch-openai-gpt6-sol-luna-launch","modelId":"gpt-6-luna","n":null,"observedOn":"2026-09-22","rawPayloadHash":null,"report":{"category":"coding","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Reported reasoning effort medium. FrontierCode 1.1 Main, as identified in the surrounding article. The chart title is FrontierCode. Graded on correctness and mergeability. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page.","family":"FrontierCode 1.1 Main (score)","locator":"Chart: FrontierCode / GPT-6 Luna / medium","metric":"percent","notes":"Provider-published lab self-report. Vega-Lite chart data value 0.3553, stored as 35.53 percent. Not an independent board.","sourceId":"openai-gpt6-sol-luna-launch","sourceTitle":"Introducing GPT-6 Sol and Luna"},"score":35.53,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-sol-and-luna/"},{"benchmarkId":"frontiercode-1-1-main-score-1-1-main","evidenceKind":"lab_self_report","harnessId":"frontiercode-1-1-main-score-1-1-main:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBoaWdoLiBGcm9udGllckNvZGUgMS4xIE1haW4sIGFzIGlkZW50aWZpZWQgaW4gdGhlIHN1cnJvdW5kaW5nIGFydGljbGUuIFRoZSBjaGFydCB0aXRsZSBpcyBGcm9udGllckNvZGUuIEdyYWRlZCBvbiBjb3JyZWN0bmVzcyBhbmQgbWVyZ2VhYmlsaXR5LiBPcGVuQUktcHVibGlzaGVkIGNoYXJ0IG9uIHRoZSBJbnRyb2R1Y2luZyBHUFQtNiBTb2wgYW5kIEx1bmEgcGFnZS4","id":"launch-openai-gpt6-sol-luna-launch-2328","ingestRunId":"launch-openai-gpt6-sol-luna-launch","modelId":"gpt-6-luna","n":null,"observedOn":"2026-09-22","rawPayloadHash":null,"report":{"category":"coding","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Reported reasoning effort high. FrontierCode 1.1 Main, as identified in the surrounding article. The chart title is FrontierCode. Graded on correctness and mergeability. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page.","family":"FrontierCode 1.1 Main (score)","locator":"Chart: FrontierCode / GPT-6 Luna / high","metric":"percent","notes":"Provider-published lab self-report. Vega-Lite chart data value 0.3726, stored as 37.26 percent. Not an independent board.","sourceId":"openai-gpt6-sol-luna-launch","sourceTitle":"Introducing GPT-6 Sol and Luna"},"score":37.26,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-sol-and-luna/"},{"benchmarkId":"frontiercode-1-1-main-score-1-1-main","evidenceKind":"lab_self_report","harnessId":"frontiercode-1-1-main-score-1-1-main:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCB4aGlnaC4gRnJvbnRpZXJDb2RlIDEuMSBNYWluLCBhcyBpZGVudGlmaWVkIGluIHRoZSBzdXJyb3VuZGluZyBhcnRpY2xlLiBUaGUgY2hhcnQgdGl0bGUgaXMgRnJvbnRpZXJDb2RlLiBHcmFkZWQgb24gY29ycmVjdG5lc3MgYW5kIG1lcmdlYWJpbGl0eS4gT3BlbkFJLXB1Ymxpc2hlZCBjaGFydCBvbiB0aGUgSW50cm9kdWNpbmcgR1BULTYgU29sIGFuZCBMdW5hIHBhZ2Uu","id":"launch-openai-gpt6-sol-luna-launch-2329","ingestRunId":"launch-openai-gpt6-sol-luna-launch","modelId":"gpt-6-luna","n":null,"observedOn":"2026-09-22","rawPayloadHash":null,"report":{"category":"coding","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Reported reasoning effort xhigh. FrontierCode 1.1 Main, as identified in the surrounding article. The chart title is FrontierCode. Graded on correctness and mergeability. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page.","family":"FrontierCode 1.1 Main (score)","locator":"Chart: FrontierCode / GPT-6 Luna / xhigh","metric":"percent","notes":"Provider-published lab self-report. Vega-Lite chart data value 0.371, stored as 37.1 percent. Not an independent board.","sourceId":"openai-gpt6-sol-luna-launch","sourceTitle":"Introducing GPT-6 Sol and Luna"},"score":37.1,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-sol-and-luna/"},{"benchmarkId":"frontiercode-1-1-main-score-1-1-main","evidenceKind":"lab_self_report","harnessId":"frontiercode-1-1-main-score-1-1-main:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIEZyb250aWVyQ29kZSAxLjEgTWFpbiwgYXMgaWRlbnRpZmllZCBpbiB0aGUgc3Vycm91bmRpbmcgYXJ0aWNsZS4gVGhlIGNoYXJ0IHRpdGxlIGlzIEZyb250aWVyQ29kZS4gR3JhZGVkIG9uIGNvcnJlY3RuZXNzIGFuZCBtZXJnZWFiaWxpdHkuIE9wZW5BSS1wdWJsaXNoZWQgY2hhcnQgb24gdGhlIEludHJvZHVjaW5nIEdQVC02IFNvbCBhbmQgTHVuYSBwYWdlLg","id":"launch-openai-gpt6-sol-luna-launch-2330","ingestRunId":"launch-openai-gpt6-sol-luna-launch","modelId":"gpt-6-luna","n":null,"observedOn":"2026-09-22","rawPayloadHash":null,"report":{"category":"coding","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Reported reasoning effort max. FrontierCode 1.1 Main, as identified in the surrounding article. The chart title is FrontierCode. Graded on correctness and mergeability. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page.","family":"FrontierCode 1.1 Main (score)","locator":"Chart: FrontierCode / GPT-6 Luna / max","metric":"percent","notes":"Provider-published lab self-report. Vega-Lite chart data value 0.4242, stored as 42.42 percent. Not an independent board.","sourceId":"openai-gpt6-sol-luna-launch","sourceTitle":"Introducing GPT-6 Sol and Luna"},"score":42.42,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-sol-and-luna/"},{"benchmarkId":"frontiercode-1-1-main-score-1-1-main","evidenceKind":"lab_self_report","harnessId":"frontiercode-1-1-main-score-1-1-main:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBsb3cuIEZyb250aWVyQ29kZSAxLjEgTWFpbiwgYXMgaWRlbnRpZmllZCBpbiB0aGUgc3Vycm91bmRpbmcgYXJ0aWNsZS4gVGhlIGNoYXJ0IHRpdGxlIGlzIEZyb250aWVyQ29kZS4gR3JhZGVkIG9uIGNvcnJlY3RuZXNzIGFuZCBtZXJnZWFiaWxpdHkuIE9wZW5BSS1wdWJsaXNoZWQgY2hhcnQgb24gdGhlIEludHJvZHVjaW5nIEdQVC02IFNvbCBhbmQgTHVuYSBwYWdlLg","id":"launch-openai-gpt6-sol-luna-launch-2331","ingestRunId":"launch-openai-gpt6-sol-luna-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-22","rawPayloadHash":null,"report":{"category":"coding","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Reported reasoning effort low. FrontierCode 1.1 Main, as identified in the surrounding article. The chart title is FrontierCode. Graded on correctness and mergeability. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page.","family":"FrontierCode 1.1 Main (score)","locator":"Chart: FrontierCode / GPT-6 Sol / low","metric":"percent","notes":"Provider-published lab self-report. Vega-Lite chart data value 0.3728, stored as 37.28 percent. Not an independent board.","sourceId":"openai-gpt6-sol-luna-launch","sourceTitle":"Introducing GPT-6 Sol and Luna"},"score":37.28,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-sol-and-luna/"},{"benchmarkId":"frontiercode-1-1-main-score-1-1-main","evidenceKind":"lab_self_report","harnessId":"frontiercode-1-1-main-score-1-1-main:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtZWRpdW0uIEZyb250aWVyQ29kZSAxLjEgTWFpbiwgYXMgaWRlbnRpZmllZCBpbiB0aGUgc3Vycm91bmRpbmcgYXJ0aWNsZS4gVGhlIGNoYXJ0IHRpdGxlIGlzIEZyb250aWVyQ29kZS4gR3JhZGVkIG9uIGNvcnJlY3RuZXNzIGFuZCBtZXJnZWFiaWxpdHkuIE9wZW5BSS1wdWJsaXNoZWQgY2hhcnQgb24gdGhlIEludHJvZHVjaW5nIEdQVC02IFNvbCBhbmQgTHVuYSBwYWdlLg","id":"launch-openai-gpt6-sol-luna-launch-2332","ingestRunId":"launch-openai-gpt6-sol-luna-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-22","rawPayloadHash":null,"report":{"category":"coding","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Reported reasoning effort medium. FrontierCode 1.1 Main, as identified in the surrounding article. The chart title is FrontierCode. Graded on correctness and mergeability. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page.","family":"FrontierCode 1.1 Main (score)","locator":"Chart: FrontierCode / GPT-6 Sol / medium","metric":"percent","notes":"Provider-published lab self-report. Vega-Lite chart data value 0.4592, stored as 45.92 percent. Not an independent board.","sourceId":"openai-gpt6-sol-luna-launch","sourceTitle":"Introducing GPT-6 Sol and Luna"},"score":45.92,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-sol-and-luna/"},{"benchmarkId":"frontiercode-1-1-main-score-1-1-main","evidenceKind":"lab_self_report","harnessId":"frontiercode-1-1-main-score-1-1-main:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBoaWdoLiBGcm9udGllckNvZGUgMS4xIE1haW4sIGFzIGlkZW50aWZpZWQgaW4gdGhlIHN1cnJvdW5kaW5nIGFydGljbGUuIFRoZSBjaGFydCB0aXRsZSBpcyBGcm9udGllckNvZGUuIEdyYWRlZCBvbiBjb3JyZWN0bmVzcyBhbmQgbWVyZ2VhYmlsaXR5LiBPcGVuQUktcHVibGlzaGVkIGNoYXJ0IG9uIHRoZSBJbnRyb2R1Y2luZyBHUFQtNiBTb2wgYW5kIEx1bmEgcGFnZS4","id":"launch-openai-gpt6-sol-luna-launch-2333","ingestRunId":"launch-openai-gpt6-sol-luna-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-22","rawPayloadHash":null,"report":{"category":"coding","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Reported reasoning effort high. FrontierCode 1.1 Main, as identified in the surrounding article. The chart title is FrontierCode. Graded on correctness and mergeability. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page.","family":"FrontierCode 1.1 Main (score)","locator":"Chart: FrontierCode / GPT-6 Sol / high","metric":"percent","notes":"Provider-published lab self-report. Vega-Lite chart data value 0.477, stored as 47.7 percent. Not an independent board.","sourceId":"openai-gpt6-sol-luna-launch","sourceTitle":"Introducing GPT-6 Sol and Luna"},"score":47.7,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-sol-and-luna/"},{"benchmarkId":"frontiercode-1-1-main-score-1-1-main","evidenceKind":"lab_self_report","harnessId":"frontiercode-1-1-main-score-1-1-main:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCB4aGlnaC4gRnJvbnRpZXJDb2RlIDEuMSBNYWluLCBhcyBpZGVudGlmaWVkIGluIHRoZSBzdXJyb3VuZGluZyBhcnRpY2xlLiBUaGUgY2hhcnQgdGl0bGUgaXMgRnJvbnRpZXJDb2RlLiBHcmFkZWQgb24gY29ycmVjdG5lc3MgYW5kIG1lcmdlYWJpbGl0eS4gT3BlbkFJLXB1Ymxpc2hlZCBjaGFydCBvbiB0aGUgSW50cm9kdWNpbmcgR1BULTYgU29sIGFuZCBMdW5hIHBhZ2Uu","id":"launch-openai-gpt6-sol-luna-launch-2334","ingestRunId":"launch-openai-gpt6-sol-luna-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-22","rawPayloadHash":null,"report":{"category":"coding","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Reported reasoning effort xhigh. FrontierCode 1.1 Main, as identified in the surrounding article. The chart title is FrontierCode. Graded on correctness and mergeability. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page.","family":"FrontierCode 1.1 Main (score)","locator":"Chart: FrontierCode / GPT-6 Sol / xhigh","metric":"percent","notes":"Provider-published lab self-report. Vega-Lite chart data value 0.4845, stored as 48.45 percent. Not an independent board.","sourceId":"openai-gpt6-sol-luna-launch","sourceTitle":"Introducing GPT-6 Sol and Luna"},"score":48.45,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-sol-and-luna/"},{"benchmarkId":"frontiercode-1-1-main-score-1-1-main","evidenceKind":"lab_self_report","harnessId":"frontiercode-1-1-main-score-1-1-main:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIEZyb250aWVyQ29kZSAxLjEgTWFpbiwgYXMgaWRlbnRpZmllZCBpbiB0aGUgc3Vycm91bmRpbmcgYXJ0aWNsZS4gVGhlIGNoYXJ0IHRpdGxlIGlzIEZyb250aWVyQ29kZS4gR3JhZGVkIG9uIGNvcnJlY3RuZXNzIGFuZCBtZXJnZWFiaWxpdHkuIE9wZW5BSS1wdWJsaXNoZWQgY2hhcnQgb24gdGhlIEludHJvZHVjaW5nIEdQVC02IFNvbCBhbmQgTHVuYSBwYWdlLg","id":"launch-openai-gpt6-sol-luna-launch-2335","ingestRunId":"launch-openai-gpt6-sol-luna-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-22","rawPayloadHash":null,"report":{"category":"coding","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Reported reasoning effort max. FrontierCode 1.1 Main, as identified in the surrounding article. The chart title is FrontierCode. Graded on correctness and mergeability. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page.","family":"FrontierCode 1.1 Main (score)","locator":"Chart: FrontierCode / GPT-6 Sol / max","metric":"percent","notes":"Provider-published lab self-report. Vega-Lite chart data value 0.4927, stored as 49.27 percent. Not an independent board.","sourceId":"openai-gpt6-sol-luna-launch","sourceTitle":"Introducing GPT-6 Sol and Luna"},"score":49.27,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-sol-and-luna/"},{"benchmarkId":"deepswe-v1-1-1-1-percent","evidenceKind":"lab_self_report","harnessId":"deepswe-v1-1-1-1-percent:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBsb3cuIERlZXBTV0UgdjEuMS4gT3JpZ2luYWwgbG9uZy1ob3Jpem9uIHNvZnR3YXJlLWVuZ2luZWVyaW5nIHRhc2tzLiBPcGVuQUktcHVibGlzaGVkIGNoYXJ0IG9uIHRoZSBJbnRyb2R1Y2luZyBHUFQtNiBTb2wgYW5kIEx1bmEgcGFnZS4","id":"launch-openai-gpt6-sol-luna-launch-2336","ingestRunId":"launch-openai-gpt6-sol-luna-launch","modelId":"gpt-6-luna","n":null,"observedOn":"2026-09-22","rawPayloadHash":null,"report":{"category":"coding","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Reported reasoning effort low. DeepSWE v1.1. Original long-horizon software-engineering tasks. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page.","family":"DeepSWE v1.1","locator":"Chart: DeepSWE / GPT-6 Luna / low","metric":"percent","notes":"Provider-published lab self-report. Vega-Lite chart data value 0.0243, stored as 2.43 percent. Not an independent board.","sourceId":"openai-gpt6-sol-luna-launch","sourceTitle":"Introducing GPT-6 Sol and Luna"},"score":2.43,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-sol-and-luna/"},{"benchmarkId":"deepswe-v1-1-1-1-percent","evidenceKind":"lab_self_report","harnessId":"deepswe-v1-1-1-1-percent:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtZWRpdW0uIERlZXBTV0UgdjEuMS4gT3JpZ2luYWwgbG9uZy1ob3Jpem9uIHNvZnR3YXJlLWVuZ2luZWVyaW5nIHRhc2tzLiBPcGVuQUktcHVibGlzaGVkIGNoYXJ0IG9uIHRoZSBJbnRyb2R1Y2luZyBHUFQtNiBTb2wgYW5kIEx1bmEgcGFnZS4","id":"launch-openai-gpt6-sol-luna-launch-2337","ingestRunId":"launch-openai-gpt6-sol-luna-launch","modelId":"gpt-6-luna","n":null,"observedOn":"2026-09-22","rawPayloadHash":null,"report":{"category":"coding","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Reported reasoning effort medium. DeepSWE v1.1. Original long-horizon software-engineering tasks. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page.","family":"DeepSWE v1.1","locator":"Chart: DeepSWE / GPT-6 Luna / medium","metric":"percent","notes":"Provider-published lab self-report. Vega-Lite chart data value 0.4447, stored as 44.47 percent. Not an independent board.","sourceId":"openai-gpt6-sol-luna-launch","sourceTitle":"Introducing GPT-6 Sol and Luna"},"score":44.47,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-sol-and-luna/"},{"benchmarkId":"deepswe-v1-1-1-1-percent","evidenceKind":"lab_self_report","harnessId":"deepswe-v1-1-1-1-percent:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBoaWdoLiBEZWVwU1dFIHYxLjEuIE9yaWdpbmFsIGxvbmctaG9yaXpvbiBzb2Z0d2FyZS1lbmdpbmVlcmluZyB0YXNrcy4gT3BlbkFJLXB1Ymxpc2hlZCBjaGFydCBvbiB0aGUgSW50cm9kdWNpbmcgR1BULTYgU29sIGFuZCBMdW5hIHBhZ2Uu","id":"launch-openai-gpt6-sol-luna-launch-2338","ingestRunId":"launch-openai-gpt6-sol-luna-launch","modelId":"gpt-6-luna","n":null,"observedOn":"2026-09-22","rawPayloadHash":null,"report":{"category":"coding","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Reported reasoning effort high. DeepSWE v1.1. Original long-horizon software-engineering tasks. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page.","family":"DeepSWE v1.1","locator":"Chart: DeepSWE / GPT-6 Luna / high","metric":"percent","notes":"Provider-published lab self-report. Vega-Lite chart data value 0.5929, stored as 59.29 percent. Not an independent board.","sourceId":"openai-gpt6-sol-luna-launch","sourceTitle":"Introducing GPT-6 Sol and Luna"},"score":59.29,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-sol-and-luna/"},{"benchmarkId":"deepswe-v1-1-1-1-percent","evidenceKind":"lab_self_report","harnessId":"deepswe-v1-1-1-1-percent:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCB4aGlnaC4gRGVlcFNXRSB2MS4xLiBPcmlnaW5hbCBsb25nLWhvcml6b24gc29mdHdhcmUtZW5naW5lZXJpbmcgdGFza3MuIE9wZW5BSS1wdWJsaXNoZWQgY2hhcnQgb24gdGhlIEludHJvZHVjaW5nIEdQVC02IFNvbCBhbmQgTHVuYSBwYWdlLg","id":"launch-openai-gpt6-sol-luna-launch-2339","ingestRunId":"launch-openai-gpt6-sol-luna-launch","modelId":"gpt-6-luna","n":null,"observedOn":"2026-09-22","rawPayloadHash":null,"report":{"category":"coding","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Reported reasoning effort xhigh. DeepSWE v1.1. Original long-horizon software-engineering tasks. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page.","family":"DeepSWE v1.1","locator":"Chart: DeepSWE / GPT-6 Luna / xhigh","metric":"percent","notes":"Provider-published lab self-report. Vega-Lite chart data value 0.6128, stored as 61.28 percent. Not an independent board.","sourceId":"openai-gpt6-sol-luna-launch","sourceTitle":"Introducing GPT-6 Sol and Luna"},"score":61.28,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-sol-and-luna/"},{"benchmarkId":"deepswe-v1-1-1-1-percent","evidenceKind":"lab_self_report","harnessId":"deepswe-v1-1-1-1-percent:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIERlZXBTV0UgdjEuMS4gT3JpZ2luYWwgbG9uZy1ob3Jpem9uIHNvZnR3YXJlLWVuZ2luZWVyaW5nIHRhc2tzLiBPcGVuQUktcHVibGlzaGVkIGNoYXJ0IG9uIHRoZSBJbnRyb2R1Y2luZyBHUFQtNiBTb2wgYW5kIEx1bmEgcGFnZS4","id":"launch-openai-gpt6-sol-luna-launch-2340","ingestRunId":"launch-openai-gpt6-sol-luna-launch","modelId":"gpt-6-luna","n":null,"observedOn":"2026-09-22","rawPayloadHash":null,"report":{"category":"coding","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Reported reasoning effort max. DeepSWE v1.1. Original long-horizon software-engineering tasks. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page.","family":"DeepSWE v1.1","locator":"Chart: DeepSWE / GPT-6 Luna / max","metric":"percent","notes":"Provider-published lab self-report. Vega-Lite chart data value 0.6659, stored as 66.59 percent. Not an independent board. Article prose rounds this max point to 66.6%.","sourceId":"openai-gpt6-sol-luna-launch","sourceTitle":"Introducing GPT-6 Sol and Luna"},"score":66.59,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-sol-and-luna/"},{"benchmarkId":"deepswe-v1-1-1-1-percent","evidenceKind":"lab_self_report","harnessId":"deepswe-v1-1-1-1-percent:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBsb3cuIERlZXBTV0UgdjEuMS4gT3JpZ2luYWwgbG9uZy1ob3Jpem9uIHNvZnR3YXJlLWVuZ2luZWVyaW5nIHRhc2tzLiBPcGVuQUktcHVibGlzaGVkIGNoYXJ0IG9uIHRoZSBJbnRyb2R1Y2luZyBHUFQtNiBTb2wgYW5kIEx1bmEgcGFnZS4","id":"launch-openai-gpt6-sol-luna-launch-2341","ingestRunId":"launch-openai-gpt6-sol-luna-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-22","rawPayloadHash":null,"report":{"category":"coding","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Reported reasoning effort low. DeepSWE v1.1. Original long-horizon software-engineering tasks. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page.","family":"DeepSWE v1.1","locator":"Chart: DeepSWE / GPT-6 Sol / low","metric":"percent","notes":"Provider-published lab self-report. Vega-Lite chart data value 0.3717, stored as 37.17 percent. Not an independent board.","sourceId":"openai-gpt6-sol-luna-launch","sourceTitle":"Introducing GPT-6 Sol and Luna"},"score":37.17,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-sol-and-luna/"},{"benchmarkId":"deepswe-v1-1-1-1-percent","evidenceKind":"lab_self_report","harnessId":"deepswe-v1-1-1-1-percent:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtZWRpdW0uIERlZXBTV0UgdjEuMS4gT3JpZ2luYWwgbG9uZy1ob3Jpem9uIHNvZnR3YXJlLWVuZ2luZWVyaW5nIHRhc2tzLiBPcGVuQUktcHVibGlzaGVkIGNoYXJ0IG9uIHRoZSBJbnRyb2R1Y2luZyBHUFQtNiBTb2wgYW5kIEx1bmEgcGFnZS4","id":"launch-openai-gpt6-sol-luna-launch-2342","ingestRunId":"launch-openai-gpt6-sol-luna-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-22","rawPayloadHash":null,"report":{"category":"coding","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Reported reasoning effort medium. DeepSWE v1.1. Original long-horizon software-engineering tasks. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page.","family":"DeepSWE v1.1","locator":"Chart: DeepSWE / GPT-6 Sol / medium","metric":"percent","notes":"Provider-published lab self-report. Vega-Lite chart data value 0.5664, stored as 56.64 percent. Not an independent board.","sourceId":"openai-gpt6-sol-luna-launch","sourceTitle":"Introducing GPT-6 Sol and Luna"},"score":56.64,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-sol-and-luna/"},{"benchmarkId":"deepswe-v1-1-1-1-percent","evidenceKind":"lab_self_report","harnessId":"deepswe-v1-1-1-1-percent:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBoaWdoLiBEZWVwU1dFIHYxLjEuIE9yaWdpbmFsIGxvbmctaG9yaXpvbiBzb2Z0d2FyZS1lbmdpbmVlcmluZyB0YXNrcy4gT3BlbkFJLXB1Ymxpc2hlZCBjaGFydCBvbiB0aGUgSW50cm9kdWNpbmcgR1BULTYgU29sIGFuZCBMdW5hIHBhZ2Uu","id":"launch-openai-gpt6-sol-luna-launch-2343","ingestRunId":"launch-openai-gpt6-sol-luna-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-22","rawPayloadHash":null,"report":{"category":"coding","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Reported reasoning effort high. DeepSWE v1.1. Original long-horizon software-engineering tasks. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page.","family":"DeepSWE v1.1","locator":"Chart: DeepSWE / GPT-6 Sol / high","metric":"percent","notes":"Provider-published lab self-report. Vega-Lite chart data value 0.6527, stored as 65.27 percent. Not an independent board.","sourceId":"openai-gpt6-sol-luna-launch","sourceTitle":"Introducing GPT-6 Sol and Luna"},"score":65.27,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-sol-and-luna/"},{"benchmarkId":"deepswe-v1-1-1-1-percent","evidenceKind":"lab_self_report","harnessId":"deepswe-v1-1-1-1-percent:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCB4aGlnaC4gRGVlcFNXRSB2MS4xLiBPcmlnaW5hbCBsb25nLWhvcml6b24gc29mdHdhcmUtZW5naW5lZXJpbmcgdGFza3MuIE9wZW5BSS1wdWJsaXNoZWQgY2hhcnQgb24gdGhlIEludHJvZHVjaW5nIEdQVC02IFNvbCBhbmQgTHVuYSBwYWdlLg","id":"launch-openai-gpt6-sol-luna-launch-2344","ingestRunId":"launch-openai-gpt6-sol-luna-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-22","rawPayloadHash":null,"report":{"category":"coding","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Reported reasoning effort xhigh. DeepSWE v1.1. Original long-horizon software-engineering tasks. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page.","family":"DeepSWE v1.1","locator":"Chart: DeepSWE / GPT-6 Sol / xhigh","metric":"percent","notes":"Provider-published lab self-report. Vega-Lite chart data value 0.6659, stored as 66.59 percent. Not an independent board.","sourceId":"openai-gpt6-sol-luna-launch","sourceTitle":"Introducing GPT-6 Sol and Luna"},"score":66.59,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-sol-and-luna/"},{"benchmarkId":"deepswe-v1-1-1-1-percent","evidenceKind":"lab_self_report","harnessId":"deepswe-v1-1-1-1-percent:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIERlZXBTV0UgdjEuMS4gT3JpZ2luYWwgbG9uZy1ob3Jpem9uIHNvZnR3YXJlLWVuZ2luZWVyaW5nIHRhc2tzLiBPcGVuQUktcHVibGlzaGVkIGNoYXJ0IG9uIHRoZSBJbnRyb2R1Y2luZyBHUFQtNiBTb2wgYW5kIEx1bmEgcGFnZS4","id":"launch-openai-gpt6-sol-luna-launch-2345","ingestRunId":"launch-openai-gpt6-sol-luna-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-22","rawPayloadHash":null,"report":{"category":"coding","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Reported reasoning effort max. DeepSWE v1.1. Original long-horizon software-engineering tasks. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page.","family":"DeepSWE v1.1","locator":"Chart: DeepSWE / GPT-6 Sol / max","metric":"percent","notes":"Provider-published lab self-report. Vega-Lite chart data value 0.6881, stored as 68.81 percent. Not an independent board. Article prose rounds this max point to 68.8%.","sourceId":"openai-gpt6-sol-luna-launch","sourceTitle":"Introducing GPT-6 Sol and Luna"},"score":68.81,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-sol-and-luna/"},{"benchmarkId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline","evidenceKind":"lab_self_report","harnessId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBsb3cuIFBhcnRpYWwgcmV3YXJkIG9uIHRoZSBvZmZsaW5lIHNldCBmcm9tIHRoZSB2MjAyNi4wOC4wOCByZWxlYXNlLiBPcGVuQUktcHVibGlzaGVkIGNoYXJ0IG9uIHRoZSBJbnRyb2R1Y2luZyBHUFQtNiBTb2wgYW5kIEx1bmEgcGFnZS4","id":"launch-openai-gpt6-sol-luna-launch-2346","ingestRunId":"launch-openai-gpt6-sol-luna-launch","modelId":"gpt-6-luna","n":null,"observedOn":"2026-09-22","rawPayloadHash":null,"report":{"category":"agentic","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Reported reasoning effort low. Partial reward on the offline set from the v2026.08.08 release. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page.","family":"OSWorld","locator":"Chart: OSWorld 2.0, offline set / GPT-6 Luna / low","metric":"percent","notes":"Provider-published lab self-report. Vega-Lite chart data value 0.0826, stored as 8.26 percent. Not an independent board.","sourceId":"openai-gpt6-sol-luna-launch","sourceTitle":"Introducing GPT-6 Sol and Luna"},"score":8.26,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-sol-and-luna/"},{"benchmarkId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline","evidenceKind":"lab_self_report","harnessId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtZWRpdW0uIFBhcnRpYWwgcmV3YXJkIG9uIHRoZSBvZmZsaW5lIHNldCBmcm9tIHRoZSB2MjAyNi4wOC4wOCByZWxlYXNlLiBPcGVuQUktcHVibGlzaGVkIGNoYXJ0IG9uIHRoZSBJbnRyb2R1Y2luZyBHUFQtNiBTb2wgYW5kIEx1bmEgcGFnZS4","id":"launch-openai-gpt6-sol-luna-launch-2347","ingestRunId":"launch-openai-gpt6-sol-luna-launch","modelId":"gpt-6-luna","n":null,"observedOn":"2026-09-22","rawPayloadHash":null,"report":{"category":"agentic","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Reported reasoning effort medium. Partial reward on the offline set from the v2026.08.08 release. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page.","family":"OSWorld","locator":"Chart: OSWorld 2.0, offline set / GPT-6 Luna / medium","metric":"percent","notes":"Provider-published lab self-report. Vega-Lite chart data value 0.3154, stored as 31.54 percent. Not an independent board.","sourceId":"openai-gpt6-sol-luna-launch","sourceTitle":"Introducing GPT-6 Sol and Luna"},"score":31.54,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-sol-and-luna/"},{"benchmarkId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline","evidenceKind":"lab_self_report","harnessId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBoaWdoLiBQYXJ0aWFsIHJld2FyZCBvbiB0aGUgb2ZmbGluZSBzZXQgZnJvbSB0aGUgdjIwMjYuMDguMDggcmVsZWFzZS4gT3BlbkFJLXB1Ymxpc2hlZCBjaGFydCBvbiB0aGUgSW50cm9kdWNpbmcgR1BULTYgU29sIGFuZCBMdW5hIHBhZ2Uu","id":"launch-openai-gpt6-sol-luna-launch-2348","ingestRunId":"launch-openai-gpt6-sol-luna-launch","modelId":"gpt-6-luna","n":null,"observedOn":"2026-09-22","rawPayloadHash":null,"report":{"category":"agentic","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Reported reasoning effort high. Partial reward on the offline set from the v2026.08.08 release. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page.","family":"OSWorld","locator":"Chart: OSWorld 2.0, offline set / GPT-6 Luna / high","metric":"percent","notes":"Provider-published lab self-report. Vega-Lite chart data value 0.4143, stored as 41.43 percent. Not an independent board.","sourceId":"openai-gpt6-sol-luna-launch","sourceTitle":"Introducing GPT-6 Sol and Luna"},"score":41.43,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-sol-and-luna/"},{"benchmarkId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline","evidenceKind":"lab_self_report","harnessId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCB4aGlnaC4gUGFydGlhbCByZXdhcmQgb24gdGhlIG9mZmxpbmUgc2V0IGZyb20gdGhlIHYyMDI2LjA4LjA4IHJlbGVhc2UuIE9wZW5BSS1wdWJsaXNoZWQgY2hhcnQgb24gdGhlIEludHJvZHVjaW5nIEdQVC02IFNvbCBhbmQgTHVuYSBwYWdlLg","id":"launch-openai-gpt6-sol-luna-launch-2349","ingestRunId":"launch-openai-gpt6-sol-luna-launch","modelId":"gpt-6-luna","n":null,"observedOn":"2026-09-22","rawPayloadHash":null,"report":{"category":"agentic","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Reported reasoning effort xhigh. Partial reward on the offline set from the v2026.08.08 release. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page.","family":"OSWorld","locator":"Chart: OSWorld 2.0, offline set / GPT-6 Luna / xhigh","metric":"percent","notes":"Provider-published lab self-report. Vega-Lite chart data value 0.467, stored as 46.7 percent. Not an independent board.","sourceId":"openai-gpt6-sol-luna-launch","sourceTitle":"Introducing GPT-6 Sol and Luna"},"score":46.7,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-sol-and-luna/"},{"benchmarkId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline","evidenceKind":"lab_self_report","harnessId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIFBhcnRpYWwgcmV3YXJkIG9uIHRoZSBvZmZsaW5lIHNldCBmcm9tIHRoZSB2MjAyNi4wOC4wOCByZWxlYXNlLiBPcGVuQUktcHVibGlzaGVkIGNoYXJ0IG9uIHRoZSBJbnRyb2R1Y2luZyBHUFQtNiBTb2wgYW5kIEx1bmEgcGFnZS4","id":"launch-openai-gpt6-sol-luna-launch-2350","ingestRunId":"launch-openai-gpt6-sol-luna-launch","modelId":"gpt-6-luna","n":null,"observedOn":"2026-09-22","rawPayloadHash":null,"report":{"category":"agentic","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Reported reasoning effort max. Partial reward on the offline set from the v2026.08.08 release. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page.","family":"OSWorld","locator":"Chart: OSWorld 2.0, offline set / GPT-6 Luna / max","metric":"percent","notes":"Provider-published lab self-report. Vega-Lite chart data value 0.5268, stored as 52.68 percent. Not an independent board.","sourceId":"openai-gpt6-sol-luna-launch","sourceTitle":"Introducing GPT-6 Sol and Luna"},"score":52.68,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-sol-and-luna/"},{"benchmarkId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline","evidenceKind":"lab_self_report","harnessId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBsb3cuIFBhcnRpYWwgcmV3YXJkIG9uIHRoZSBvZmZsaW5lIHNldCBmcm9tIHRoZSB2MjAyNi4wOC4wOCByZWxlYXNlLiBPcGVuQUktcHVibGlzaGVkIGNoYXJ0IG9uIHRoZSBJbnRyb2R1Y2luZyBHUFQtNiBTb2wgYW5kIEx1bmEgcGFnZS4","id":"launch-openai-gpt6-sol-luna-launch-2351","ingestRunId":"launch-openai-gpt6-sol-luna-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-22","rawPayloadHash":null,"report":{"category":"agentic","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Reported reasoning effort low. Partial reward on the offline set from the v2026.08.08 release. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page.","family":"OSWorld","locator":"Chart: OSWorld 2.0, offline set / GPT-6 Sol / low","metric":"percent","notes":"Provider-published lab self-report. Vega-Lite chart data value 0.439, stored as 43.9 percent. Not an independent board.","sourceId":"openai-gpt6-sol-luna-launch","sourceTitle":"Introducing GPT-6 Sol and Luna"},"score":43.9,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-sol-and-luna/"},{"benchmarkId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline","evidenceKind":"lab_self_report","harnessId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtZWRpdW0uIFBhcnRpYWwgcmV3YXJkIG9uIHRoZSBvZmZsaW5lIHNldCBmcm9tIHRoZSB2MjAyNi4wOC4wOCByZWxlYXNlLiBPcGVuQUktcHVibGlzaGVkIGNoYXJ0IG9uIHRoZSBJbnRyb2R1Y2luZyBHUFQtNiBTb2wgYW5kIEx1bmEgcGFnZS4","id":"launch-openai-gpt6-sol-luna-launch-2352","ingestRunId":"launch-openai-gpt6-sol-luna-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-22","rawPayloadHash":null,"report":{"category":"agentic","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Reported reasoning effort medium. Partial reward on the offline set from the v2026.08.08 release. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page.","family":"OSWorld","locator":"Chart: OSWorld 2.0, offline set / GPT-6 Sol / medium","metric":"percent","notes":"Provider-published lab self-report. Vega-Lite chart data value 0.54, stored as 54 percent. Not an independent board.","sourceId":"openai-gpt6-sol-luna-launch","sourceTitle":"Introducing GPT-6 Sol and Luna"},"score":54,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-sol-and-luna/"},{"benchmarkId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline","evidenceKind":"lab_self_report","harnessId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBoaWdoLiBQYXJ0aWFsIHJld2FyZCBvbiB0aGUgb2ZmbGluZSBzZXQgZnJvbSB0aGUgdjIwMjYuMDguMDggcmVsZWFzZS4gT3BlbkFJLXB1Ymxpc2hlZCBjaGFydCBvbiB0aGUgSW50cm9kdWNpbmcgR1BULTYgU29sIGFuZCBMdW5hIHBhZ2Uu","id":"launch-openai-gpt6-sol-luna-launch-2353","ingestRunId":"launch-openai-gpt6-sol-luna-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-22","rawPayloadHash":null,"report":{"category":"agentic","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Reported reasoning effort high. Partial reward on the offline set from the v2026.08.08 release. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page.","family":"OSWorld","locator":"Chart: OSWorld 2.0, offline set / GPT-6 Sol / high","metric":"percent","notes":"Provider-published lab self-report. Vega-Lite chart data value 0.5829, stored as 58.29 percent. Not an independent board.","sourceId":"openai-gpt6-sol-luna-launch","sourceTitle":"Introducing GPT-6 Sol and Luna"},"score":58.29,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-sol-and-luna/"},{"benchmarkId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline","evidenceKind":"lab_self_report","harnessId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCB4aGlnaC4gUGFydGlhbCByZXdhcmQgb24gdGhlIG9mZmxpbmUgc2V0IGZyb20gdGhlIHYyMDI2LjA4LjA4IHJlbGVhc2UuIE9wZW5BSS1wdWJsaXNoZWQgY2hhcnQgb24gdGhlIEludHJvZHVjaW5nIEdQVC02IFNvbCBhbmQgTHVuYSBwYWdlLg","id":"launch-openai-gpt6-sol-luna-launch-2354","ingestRunId":"launch-openai-gpt6-sol-luna-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-22","rawPayloadHash":null,"report":{"category":"agentic","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Reported reasoning effort xhigh. Partial reward on the offline set from the v2026.08.08 release. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page.","family":"OSWorld","locator":"Chart: OSWorld 2.0, offline set / GPT-6 Sol / xhigh","metric":"percent","notes":"Provider-published lab self-report. Vega-Lite chart data value 0.6054, stored as 60.54 percent. Not an independent board. Article prose rounds this xhigh point to 60.5%.","sourceId":"openai-gpt6-sol-luna-launch","sourceTitle":"Introducing GPT-6 Sol and Luna"},"score":60.54,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-sol-and-luna/"},{"benchmarkId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline","evidenceKind":"lab_self_report","harnessId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline:openai-gpt6-sol-luna-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIFBhcnRpYWwgcmV3YXJkIG9uIHRoZSBvZmZsaW5lIHNldCBmcm9tIHRoZSB2MjAyNi4wOC4wOCByZWxlYXNlLiBPcGVuQUktcHVibGlzaGVkIGNoYXJ0IG9uIHRoZSBJbnRyb2R1Y2luZyBHUFQtNiBTb2wgYW5kIEx1bmEgcGFnZS4","id":"launch-openai-gpt6-sol-luna-launch-2355","ingestRunId":"launch-openai-gpt6-sol-luna-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-22","rawPayloadHash":null,"report":{"category":"agentic","comparabilityReason":"Provider-published lab self-report. Shown as a labeled claim and not admitted as a matched board comparison.","comparable":false,"configuration":"Reported reasoning effort max. Partial reward on the offline set from the v2026.08.08 release. OpenAI-published chart on the Introducing GPT-6 Sol and Luna page.","family":"OSWorld","locator":"Chart: OSWorld 2.0, offline set / GPT-6 Sol / max","metric":"percent","notes":"Provider-published lab self-report. Vega-Lite chart data value 0.6443, stored as 64.43 percent. Not an independent board.","sourceId":"openai-gpt6-sol-luna-launch","sourceTitle":"Introducing GPT-6 Sol and Luna"},"score":64.43,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-sol-and-luna/"},{"benchmarkId":"deepswe-v1-1-1-1-percent","evidenceKind":"lab_self_report","harnessId":"deepswe-v1-1-1-1-percent:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBsb3cuIENvbXBsZXggc29mdHdhcmUtZW5naW5lZXJpbmcgdGFza3MgaW4gb3JpZ2luYWwgcmVhbCBjb2RlYmFzZXMuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","id":"launch-openai-gpt61-sol-launch-2356","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6.1-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"coding","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort low. Complex software-engineering tasks in original real codebases. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"DeepSWE v1.1","locator":"Chart: DeepSWE / GPT-6.1 Sol / low","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $0.1714; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":64.38,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"deepswe-v1-1-1-1-percent","evidenceKind":"lab_self_report","harnessId":"deepswe-v1-1-1-1-percent:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBsb3cuIENvbXBsZXggc29mdHdhcmUtZW5naW5lZXJpbmcgdGFza3MgaW4gb3JpZ2luYWwgcmVhbCBjb2RlYmFzZXMuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","id":"launch-openai-gpt61-sol-launch-2357","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"coding","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort low. Complex software-engineering tasks in original real codebases. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"DeepSWE v1.1","locator":"Chart: DeepSWE / GPT-6 Sol / low","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $0.1623; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":37.17,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"deepswe-v1-1-1-1-percent","evidenceKind":"lab_self_report","harnessId":"deepswe-v1-1-1-1-percent:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBsb3cuIENvbXBsZXggc29mdHdhcmUtZW5naW5lZXJpbmcgdGFza3MgaW4gb3JpZ2luYWwgcmVhbCBjb2RlYmFzZXMuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","id":"launch-openai-gpt61-sol-launch-2358","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"coding","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort low. Complex software-engineering tasks in original real codebases. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"DeepSWE v1.1","locator":"Chart: DeepSWE / GPT-6 Astra / low","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $1.5952; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":67.04,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"deepswe-v1-1-1-1-percent","evidenceKind":"lab_self_report","harnessId":"deepswe-v1-1-1-1-percent:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtZWRpdW0uIENvbXBsZXggc29mdHdhcmUtZW5naW5lZXJpbmcgdGFza3MgaW4gb3JpZ2luYWwgcmVhbCBjb2RlYmFzZXMuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","id":"launch-openai-gpt61-sol-launch-2359","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6.1-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"coding","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort medium. Complex software-engineering tasks in original real codebases. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"DeepSWE v1.1","locator":"Chart: DeepSWE / GPT-6.1 Sol / medium","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $0.4196; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":73.01,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"deepswe-v1-1-1-1-percent","evidenceKind":"lab_self_report","harnessId":"deepswe-v1-1-1-1-percent:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtZWRpdW0uIENvbXBsZXggc29mdHdhcmUtZW5naW5lZXJpbmcgdGFza3MgaW4gb3JpZ2luYWwgcmVhbCBjb2RlYmFzZXMuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","id":"launch-openai-gpt61-sol-launch-2360","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"coding","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort medium. Complex software-engineering tasks in original real codebases. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"DeepSWE v1.1","locator":"Chart: DeepSWE / GPT-6 Sol / medium","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $0.3798; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":56.64,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"deepswe-v1-1-1-1-percent","evidenceKind":"lab_self_report","harnessId":"deepswe-v1-1-1-1-percent:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtZWRpdW0uIENvbXBsZXggc29mdHdhcmUtZW5naW5lZXJpbmcgdGFza3MgaW4gb3JpZ2luYWwgcmVhbCBjb2RlYmFzZXMuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","id":"launch-openai-gpt61-sol-launch-2361","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"coding","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort medium. Complex software-engineering tasks in original real codebases. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"DeepSWE v1.1","locator":"Chart: DeepSWE / GPT-6 Astra / medium","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $3.0755; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":72.79,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"deepswe-v1-1-1-1-percent","evidenceKind":"lab_self_report","harnessId":"deepswe-v1-1-1-1-percent:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBoaWdoLiBDb21wbGV4IHNvZnR3YXJlLWVuZ2luZWVyaW5nIHRhc2tzIGluIG9yaWdpbmFsIHJlYWwgY29kZWJhc2VzLiBPcGVuQUkgcmVzZWFyY2ggZW52aXJvbm1lbnQgb3IgQVBJOyBzYW1lIG5hbWVkIG1ldHJpYyBhbmQgZXZhbHVhdGlvbiBjaGFydC4gTm8gZmFsbGJhY2sgaXMgcmVwb3J0ZWQgZm9yIHRoaXMgT3BlbkFJIGNvbmZpZ3VyYXRpb24u","id":"launch-openai-gpt61-sol-launch-2362","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6.1-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"coding","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort high. Complex software-engineering tasks in original real codebases. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"DeepSWE v1.1","locator":"Chart: DeepSWE / GPT-6.1 Sol / high","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $0.6461; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":75.22,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"deepswe-v1-1-1-1-percent","evidenceKind":"lab_self_report","harnessId":"deepswe-v1-1-1-1-percent:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBoaWdoLiBDb21wbGV4IHNvZnR3YXJlLWVuZ2luZWVyaW5nIHRhc2tzIGluIG9yaWdpbmFsIHJlYWwgY29kZWJhc2VzLiBPcGVuQUkgcmVzZWFyY2ggZW52aXJvbm1lbnQgb3IgQVBJOyBzYW1lIG5hbWVkIG1ldHJpYyBhbmQgZXZhbHVhdGlvbiBjaGFydC4gTm8gZmFsbGJhY2sgaXMgcmVwb3J0ZWQgZm9yIHRoaXMgT3BlbkFJIGNvbmZpZ3VyYXRpb24u","id":"launch-openai-gpt61-sol-launch-2363","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"coding","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort high. Complex software-engineering tasks in original real codebases. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"DeepSWE v1.1","locator":"Chart: DeepSWE / GPT-6 Sol / high","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $0.6404; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":65.27,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"deepswe-v1-1-1-1-percent","evidenceKind":"lab_self_report","harnessId":"deepswe-v1-1-1-1-percent:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBoaWdoLiBDb21wbGV4IHNvZnR3YXJlLWVuZ2luZWVyaW5nIHRhc2tzIGluIG9yaWdpbmFsIHJlYWwgY29kZWJhc2VzLiBPcGVuQUkgcmVzZWFyY2ggZW52aXJvbm1lbnQgb3IgQVBJOyBzYW1lIG5hbWVkIG1ldHJpYyBhbmQgZXZhbHVhdGlvbiBjaGFydC4gTm8gZmFsbGJhY2sgaXMgcmVwb3J0ZWQgZm9yIHRoaXMgT3BlbkFJIGNvbmZpZ3VyYXRpb24u","id":"launch-openai-gpt61-sol-launch-2364","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"coding","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort high. Complex software-engineering tasks in original real codebases. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"DeepSWE v1.1","locator":"Chart: DeepSWE / GPT-6 Astra / high","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $3.9237; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":73.23,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"deepswe-v1-1-1-1-percent","evidenceKind":"lab_self_report","harnessId":"deepswe-v1-1-1-1-percent:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCB4aGlnaC4gQ29tcGxleCBzb2Z0d2FyZS1lbmdpbmVlcmluZyB0YXNrcyBpbiBvcmlnaW5hbCByZWFsIGNvZGViYXNlcy4gT3BlbkFJIHJlc2VhcmNoIGVudmlyb25tZW50IG9yIEFQSTsgc2FtZSBuYW1lZCBtZXRyaWMgYW5kIGV2YWx1YXRpb24gY2hhcnQuIE5vIGZhbGxiYWNrIGlzIHJlcG9ydGVkIGZvciB0aGlzIE9wZW5BSSBjb25maWd1cmF0aW9uLg","id":"launch-openai-gpt61-sol-launch-2365","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6.1-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"coding","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort xhigh. Complex software-engineering tasks in original real codebases. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"DeepSWE v1.1","locator":"Chart: DeepSWE / GPT-6.1 Sol / xhigh","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $0.7886; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":71.9,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"deepswe-v1-1-1-1-percent","evidenceKind":"lab_self_report","harnessId":"deepswe-v1-1-1-1-percent:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCB4aGlnaC4gQ29tcGxleCBzb2Z0d2FyZS1lbmdpbmVlcmluZyB0YXNrcyBpbiBvcmlnaW5hbCByZWFsIGNvZGViYXNlcy4gT3BlbkFJIHJlc2VhcmNoIGVudmlyb25tZW50IG9yIEFQSTsgc2FtZSBuYW1lZCBtZXRyaWMgYW5kIGV2YWx1YXRpb24gY2hhcnQuIE5vIGZhbGxiYWNrIGlzIHJlcG9ydGVkIGZvciB0aGlzIE9wZW5BSSBjb25maWd1cmF0aW9uLg","id":"launch-openai-gpt61-sol-launch-2366","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"coding","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort xhigh. Complex software-engineering tasks in original real codebases. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"DeepSWE v1.1","locator":"Chart: DeepSWE / GPT-6 Sol / xhigh","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $1.0033; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":66.59,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"deepswe-v1-1-1-1-percent","evidenceKind":"lab_self_report","harnessId":"deepswe-v1-1-1-1-percent:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCB4aGlnaC4gQ29tcGxleCBzb2Z0d2FyZS1lbmdpbmVlcmluZyB0YXNrcyBpbiBvcmlnaW5hbCByZWFsIGNvZGViYXNlcy4gT3BlbkFJIHJlc2VhcmNoIGVudmlyb25tZW50IG9yIEFQSTsgc2FtZSBuYW1lZCBtZXRyaWMgYW5kIGV2YWx1YXRpb24gY2hhcnQuIE5vIGZhbGxiYWNrIGlzIHJlcG9ydGVkIGZvciB0aGlzIE9wZW5BSSBjb25maWd1cmF0aW9uLg","id":"launch-openai-gpt61-sol-launch-2367","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"coding","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort xhigh. Complex software-engineering tasks in original real codebases. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"DeepSWE v1.1","locator":"Chart: DeepSWE / GPT-6 Astra / xhigh","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $4.4291; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":74.12,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"deepswe-v1-1-1-1-percent","evidenceKind":"lab_self_report","harnessId":"deepswe-v1-1-1-1-percent:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIENvbXBsZXggc29mdHdhcmUtZW5naW5lZXJpbmcgdGFza3MgaW4gb3JpZ2luYWwgcmVhbCBjb2RlYmFzZXMuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","id":"launch-openai-gpt61-sol-launch-2368","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6.1-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"coding","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort max. Complex software-engineering tasks in original real codebases. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"DeepSWE v1.1","locator":"Chart: DeepSWE / GPT-6.1 Sol / max","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $1.5711; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":71.9,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"deepswe-v1-1-1-1-percent","evidenceKind":"lab_self_report","harnessId":"deepswe-v1-1-1-1-percent:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIENvbXBsZXggc29mdHdhcmUtZW5naW5lZXJpbmcgdGFza3MgaW4gb3JpZ2luYWwgcmVhbCBjb2RlYmFzZXMuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","id":"launch-openai-gpt61-sol-launch-2369","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"coding","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort max. Complex software-engineering tasks in original real codebases. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"DeepSWE v1.1","locator":"Chart: DeepSWE / GPT-6 Sol / max","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $2.7439; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":68.81,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"deepswe-v1-1-1-1-percent","evidenceKind":"lab_self_report","harnessId":"deepswe-v1-1-1-1-percent:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIENvbXBsZXggc29mdHdhcmUtZW5naW5lZXJpbmcgdGFza3MgaW4gb3JpZ2luYWwgcmVhbCBjb2RlYmFzZXMuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","id":"launch-openai-gpt61-sol-launch-2370","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"coding","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort max. Complex software-engineering tasks in original real codebases. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"DeepSWE v1.1","locator":"Chart: DeepSWE / GPT-6 Astra / max","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $7.4978; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":73.23,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"gdp-pdf-not-specified","evidenceKind":"lab_self_report","harnessId":"gdp-pdf-not-specified:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBsb3cuIFByb2Zlc3Npb25hbCBxdWVzdGlvbnMgYWJvdXQgY29tcGxleCBQREZzIGFjcm9zcyB0ZW4gcHJvZmVzc2lvbmFsIGRvbWFpbnMuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","id":"launch-openai-gpt61-sol-launch-2371","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6.1-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort low. Professional questions about complex PDFs across ten professional domains. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"gdp.pdf","locator":"Chart: GDP.pdf / GPT-6.1 Sol / low","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $0.3341; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":27,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"gdp-pdf-not-specified","evidenceKind":"lab_self_report","harnessId":"gdp-pdf-not-specified:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtZWRpdW0uIFByb2Zlc3Npb25hbCBxdWVzdGlvbnMgYWJvdXQgY29tcGxleCBQREZzIGFjcm9zcyB0ZW4gcHJvZmVzc2lvbmFsIGRvbWFpbnMuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","id":"launch-openai-gpt61-sol-launch-2372","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6.1-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort medium. Professional questions about complex PDFs across ten professional domains. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"gdp.pdf","locator":"Chart: GDP.pdf / GPT-6.1 Sol / medium","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $0.3375; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":30,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"gdp-pdf-not-specified","evidenceKind":"lab_self_report","harnessId":"gdp-pdf-not-specified:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBoaWdoLiBQcm9mZXNzaW9uYWwgcXVlc3Rpb25zIGFib3V0IGNvbXBsZXggUERGcyBhY3Jvc3MgdGVuIHByb2Zlc3Npb25hbCBkb21haW5zLiBPcGVuQUkgcmVzZWFyY2ggZW52aXJvbm1lbnQgb3IgQVBJOyBzYW1lIG5hbWVkIG1ldHJpYyBhbmQgZXZhbHVhdGlvbiBjaGFydC4gTm8gZmFsbGJhY2sgaXMgcmVwb3J0ZWQgZm9yIHRoaXMgT3BlbkFJIGNvbmZpZ3VyYXRpb24u","id":"launch-openai-gpt61-sol-launch-2373","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6.1-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort high. Professional questions about complex PDFs across ten professional domains. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"gdp.pdf","locator":"Chart: GDP.pdf / GPT-6.1 Sol / high","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $0.3494; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":32,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"gdp-pdf-not-specified","evidenceKind":"lab_self_report","harnessId":"gdp-pdf-not-specified:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCB4aGlnaC4gUHJvZmVzc2lvbmFsIHF1ZXN0aW9ucyBhYm91dCBjb21wbGV4IFBERnMgYWNyb3NzIHRlbiBwcm9mZXNzaW9uYWwgZG9tYWlucy4gT3BlbkFJIHJlc2VhcmNoIGVudmlyb25tZW50IG9yIEFQSTsgc2FtZSBuYW1lZCBtZXRyaWMgYW5kIGV2YWx1YXRpb24gY2hhcnQuIE5vIGZhbGxiYWNrIGlzIHJlcG9ydGVkIGZvciB0aGlzIE9wZW5BSSBjb25maWd1cmF0aW9uLg","id":"launch-openai-gpt61-sol-launch-2374","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6.1-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort xhigh. Professional questions about complex PDFs across ten professional domains. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"gdp.pdf","locator":"Chart: GDP.pdf / GPT-6.1 Sol / xhigh","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $0.3681; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":31.8,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"gdp-pdf-not-specified","evidenceKind":"lab_self_report","harnessId":"gdp-pdf-not-specified:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIFByb2Zlc3Npb25hbCBxdWVzdGlvbnMgYWJvdXQgY29tcGxleCBQREZzIGFjcm9zcyB0ZW4gcHJvZmVzc2lvbmFsIGRvbWFpbnMuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","id":"launch-openai-gpt61-sol-launch-2375","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6.1-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort max. Professional questions about complex PDFs across ten professional domains. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"gdp.pdf","locator":"Chart: GDP.pdf / GPT-6.1 Sol / max","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $0.4199; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":31,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"gdp-pdf-not-specified","evidenceKind":"lab_self_report","harnessId":"gdp-pdf-not-specified:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBsb3cuIFByb2Zlc3Npb25hbCBxdWVzdGlvbnMgYWJvdXQgY29tcGxleCBQREZzIGFjcm9zcyB0ZW4gcHJvZmVzc2lvbmFsIGRvbWFpbnMuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","id":"launch-openai-gpt61-sol-launch-2376","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort low. Professional questions about complex PDFs across ten professional domains. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"gdp.pdf","locator":"Chart: GDP.pdf / GPT-6 Astra / low","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $1.6963186; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":30.4,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"gdp-pdf-not-specified","evidenceKind":"lab_self_report","harnessId":"gdp-pdf-not-specified:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtZWRpdW0uIFByb2Zlc3Npb25hbCBxdWVzdGlvbnMgYWJvdXQgY29tcGxleCBQREZzIGFjcm9zcyB0ZW4gcHJvZmVzc2lvbmFsIGRvbWFpbnMuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","id":"launch-openai-gpt61-sol-launch-2377","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort medium. Professional questions about complex PDFs across ten professional domains. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"gdp.pdf","locator":"Chart: GDP.pdf / GPT-6 Astra / medium","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $1.7198086; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":30.4,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"gdp-pdf-not-specified","evidenceKind":"lab_self_report","harnessId":"gdp-pdf-not-specified:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBoaWdoLiBQcm9mZXNzaW9uYWwgcXVlc3Rpb25zIGFib3V0IGNvbXBsZXggUERGcyBhY3Jvc3MgdGVuIHByb2Zlc3Npb25hbCBkb21haW5zLiBPcGVuQUkgcmVzZWFyY2ggZW52aXJvbm1lbnQgb3IgQVBJOyBzYW1lIG5hbWVkIG1ldHJpYyBhbmQgZXZhbHVhdGlvbiBjaGFydC4gTm8gZmFsbGJhY2sgaXMgcmVwb3J0ZWQgZm9yIHRoaXMgT3BlbkFJIGNvbmZpZ3VyYXRpb24u","id":"launch-openai-gpt61-sol-launch-2378","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort high. Professional questions about complex PDFs across ten professional domains. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"gdp.pdf","locator":"Chart: GDP.pdf / GPT-6 Astra / high","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $1.7948240999999998; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":31,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"gdp-pdf-not-specified","evidenceKind":"lab_self_report","harnessId":"gdp-pdf-not-specified:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCB4aGlnaC4gUHJvZmVzc2lvbmFsIHF1ZXN0aW9ucyBhYm91dCBjb21wbGV4IFBERnMgYWNyb3NzIHRlbiBwcm9mZXNzaW9uYWwgZG9tYWlucy4gT3BlbkFJIHJlc2VhcmNoIGVudmlyb25tZW50IG9yIEFQSTsgc2FtZSBuYW1lZCBtZXRyaWMgYW5kIGV2YWx1YXRpb24gY2hhcnQuIE5vIGZhbGxiYWNrIGlzIHJlcG9ydGVkIGZvciB0aGlzIE9wZW5BSSBjb25maWd1cmF0aW9uLg","id":"launch-openai-gpt61-sol-launch-2379","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort xhigh. Professional questions about complex PDFs across ten professional domains. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"gdp.pdf","locator":"Chart: GDP.pdf / GPT-6 Astra / xhigh","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $1.9127230999999998; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":32.2,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"gdp-pdf-not-specified","evidenceKind":"lab_self_report","harnessId":"gdp-pdf-not-specified:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIFByb2Zlc3Npb25hbCBxdWVzdGlvbnMgYWJvdXQgY29tcGxleCBQREZzIGFjcm9zcyB0ZW4gcHJvZmVzc2lvbmFsIGRvbWFpbnMuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","id":"launch-openai-gpt61-sol-launch-2380","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort max. Professional questions about complex PDFs across ten professional domains. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"gdp.pdf","locator":"Chart: GDP.pdf / GPT-6 Astra / max","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $2.0758207; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":31,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"gdp-pdf-not-specified","evidenceKind":"lab_self_report","harnessId":"gdp-pdf-not-specified:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBsb3cuIFByb2Zlc3Npb25hbCBxdWVzdGlvbnMgYWJvdXQgY29tcGxleCBQREZzIGFjcm9zcyB0ZW4gcHJvZmVzc2lvbmFsIGRvbWFpbnMuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","id":"launch-openai-gpt61-sol-launch-2381","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort low. Professional questions about complex PDFs across ten professional domains. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"gdp.pdf","locator":"Chart: GDP.pdf / GPT-6 Sol / low","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $0.33; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":21.8,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"gdp-pdf-not-specified","evidenceKind":"lab_self_report","harnessId":"gdp-pdf-not-specified:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtZWRpdW0uIFByb2Zlc3Npb25hbCBxdWVzdGlvbnMgYWJvdXQgY29tcGxleCBQREZzIGFjcm9zcyB0ZW4gcHJvZmVzc2lvbmFsIGRvbWFpbnMuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","id":"launch-openai-gpt61-sol-launch-2382","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort medium. Professional questions about complex PDFs across ten professional domains. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"gdp.pdf","locator":"Chart: GDP.pdf / GPT-6 Sol / medium","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $0.34; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":25.4,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"gdp-pdf-not-specified","evidenceKind":"lab_self_report","harnessId":"gdp-pdf-not-specified:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBoaWdoLiBQcm9mZXNzaW9uYWwgcXVlc3Rpb25zIGFib3V0IGNvbXBsZXggUERGcyBhY3Jvc3MgdGVuIHByb2Zlc3Npb25hbCBkb21haW5zLiBPcGVuQUkgcmVzZWFyY2ggZW52aXJvbm1lbnQgb3IgQVBJOyBzYW1lIG5hbWVkIG1ldHJpYyBhbmQgZXZhbHVhdGlvbiBjaGFydC4gTm8gZmFsbGJhY2sgaXMgcmVwb3J0ZWQgZm9yIHRoaXMgT3BlbkFJIGNvbmZpZ3VyYXRpb24u","id":"launch-openai-gpt61-sol-launch-2383","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort high. Professional questions about complex PDFs across ten professional domains. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"gdp.pdf","locator":"Chart: GDP.pdf / GPT-6 Sol / high","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $0.35; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":28,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"gdp-pdf-not-specified","evidenceKind":"lab_self_report","harnessId":"gdp-pdf-not-specified:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCB4aGlnaC4gUHJvZmVzc2lvbmFsIHF1ZXN0aW9ucyBhYm91dCBjb21wbGV4IFBERnMgYWNyb3NzIHRlbiBwcm9mZXNzaW9uYWwgZG9tYWlucy4gT3BlbkFJIHJlc2VhcmNoIGVudmlyb25tZW50IG9yIEFQSTsgc2FtZSBuYW1lZCBtZXRyaWMgYW5kIGV2YWx1YXRpb24gY2hhcnQuIE5vIGZhbGxiYWNrIGlzIHJlcG9ydGVkIGZvciB0aGlzIE9wZW5BSSBjb25maWd1cmF0aW9uLg","id":"launch-openai-gpt61-sol-launch-2384","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort xhigh. Professional questions about complex PDFs across ten professional domains. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"gdp.pdf","locator":"Chart: GDP.pdf / GPT-6 Sol / xhigh","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $0.37; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":23.8,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"gdp-pdf-not-specified","evidenceKind":"lab_self_report","harnessId":"gdp-pdf-not-specified:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIFByb2Zlc3Npb25hbCBxdWVzdGlvbnMgYWJvdXQgY29tcGxleCBQREZzIGFjcm9zcyB0ZW4gcHJvZmVzc2lvbmFsIGRvbWFpbnMuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","id":"launch-openai-gpt61-sol-launch-2385","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort max. Professional questions about complex PDFs across ten professional domains. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"gdp.pdf","locator":"Chart: GDP.pdf / GPT-6 Sol / max","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $0.43; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":24.8,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"gdp-pdf-not-specified","evidenceKind":"lab_self_report","harnessId":"gdp-pdf-not-specified:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBsb3cuIFByb2Zlc3Npb25hbCBxdWVzdGlvbnMgYWJvdXQgY29tcGxleCBQREZzIGFjcm9zcyB0ZW4gcHJvZmVzc2lvbmFsIGRvbWFpbnMuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBSZXBvcnRlZCBjb21iaW5lZCBzZXR1cDogT3B1cyA1LjUgdy8gZmFsbGJhY2tzLiBGYWxsYmFjayBiZWhhdmlvciBtdXN0IG5vdCBiZSBhdHRyaWJ1dGVkIHRvIHRoZSBiYXNlIG1vZGVsIGFsb25lLg","id":"launch-openai-gpt61-sol-launch-2386","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Combined fallback setup is not the named base model at one effort setting.","comparable":false,"configuration":"Reported reasoning effort low. Professional questions about complex PDFs across ten professional domains. OpenAI research environment or API; same named metric and evaluation chart. Reported combined setup: Opus 5.5 w/ fallbacks. Fallback behavior must not be attributed to the base model alone.","family":"gdp.pdf","locator":"Chart: GDP.pdf / Opus 5.5 w/ fallbacks / low","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $0.76411812; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":25.6,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"gdp-pdf-not-specified","evidenceKind":"lab_self_report","harnessId":"gdp-pdf-not-specified:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtZWRpdW0uIFByb2Zlc3Npb25hbCBxdWVzdGlvbnMgYWJvdXQgY29tcGxleCBQREZzIGFjcm9zcyB0ZW4gcHJvZmVzc2lvbmFsIGRvbWFpbnMuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBSZXBvcnRlZCBjb21iaW5lZCBzZXR1cDogT3B1cyA1LjUgdy8gZmFsbGJhY2tzLiBGYWxsYmFjayBiZWhhdmlvciBtdXN0IG5vdCBiZSBhdHRyaWJ1dGVkIHRvIHRoZSBiYXNlIG1vZGVsIGFsb25lLg","id":"launch-openai-gpt61-sol-launch-2387","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Combined fallback setup is not the named base model at one effort setting.","comparable":false,"configuration":"Reported reasoning effort medium. Professional questions about complex PDFs across ten professional domains. OpenAI research environment or API; same named metric and evaluation chart. Reported combined setup: Opus 5.5 w/ fallbacks. Fallback behavior must not be attributed to the base model alone.","family":"gdp.pdf","locator":"Chart: GDP.pdf / Opus 5.5 w/ fallbacks / medium","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $0.7953176399999999; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":25.6,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"gdp-pdf-not-specified","evidenceKind":"lab_self_report","harnessId":"gdp-pdf-not-specified:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBoaWdoLiBQcm9mZXNzaW9uYWwgcXVlc3Rpb25zIGFib3V0IGNvbXBsZXggUERGcyBhY3Jvc3MgdGVuIHByb2Zlc3Npb25hbCBkb21haW5zLiBPcGVuQUkgcmVzZWFyY2ggZW52aXJvbm1lbnQgb3IgQVBJOyBzYW1lIG5hbWVkIG1ldHJpYyBhbmQgZXZhbHVhdGlvbiBjaGFydC4gUmVwb3J0ZWQgY29tYmluZWQgc2V0dXA6IE9wdXMgNS41IHcvIGZhbGxiYWNrcy4gRmFsbGJhY2sgYmVoYXZpb3IgbXVzdCBub3QgYmUgYXR0cmlidXRlZCB0byB0aGUgYmFzZSBtb2RlbCBhbG9uZS4","id":"launch-openai-gpt61-sol-launch-2388","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Combined fallback setup is not the named base model at one effort setting.","comparable":false,"configuration":"Reported reasoning effort high. Professional questions about complex PDFs across ten professional domains. OpenAI research environment or API; same named metric and evaluation chart. Reported combined setup: Opus 5.5 w/ fallbacks. Fallback behavior must not be attributed to the base model alone.","family":"gdp.pdf","locator":"Chart: GDP.pdf / Opus 5.5 w/ fallbacks / high","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $0.82542564; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":28.800000000000004,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"gdp-pdf-not-specified","evidenceKind":"lab_self_report","harnessId":"gdp-pdf-not-specified:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCB4aGlnaC4gUHJvZmVzc2lvbmFsIHF1ZXN0aW9ucyBhYm91dCBjb21wbGV4IFBERnMgYWNyb3NzIHRlbiBwcm9mZXNzaW9uYWwgZG9tYWlucy4gT3BlbkFJIHJlc2VhcmNoIGVudmlyb25tZW50IG9yIEFQSTsgc2FtZSBuYW1lZCBtZXRyaWMgYW5kIGV2YWx1YXRpb24gY2hhcnQuIFJlcG9ydGVkIGNvbWJpbmVkIHNldHVwOiBPcHVzIDUuNSB3LyBmYWxsYmFja3MuIEZhbGxiYWNrIGJlaGF2aW9yIG11c3Qgbm90IGJlIGF0dHJpYnV0ZWQgdG8gdGhlIGJhc2UgbW9kZWwgYWxvbmUu","id":"launch-openai-gpt61-sol-launch-2389","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Combined fallback setup is not the named base model at one effort setting.","comparable":false,"configuration":"Reported reasoning effort xhigh. Professional questions about complex PDFs across ten professional domains. OpenAI research environment or API; same named metric and evaluation chart. Reported combined setup: Opus 5.5 w/ fallbacks. Fallback behavior must not be attributed to the base model alone.","family":"gdp.pdf","locator":"Chart: GDP.pdf / Opus 5.5 w/ fallbacks / xhigh","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $0.9642728800000001; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":26.6,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"gdp-pdf-not-specified","evidenceKind":"lab_self_report","harnessId":"gdp-pdf-not-specified:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIFByb2Zlc3Npb25hbCBxdWVzdGlvbnMgYWJvdXQgY29tcGxleCBQREZzIGFjcm9zcyB0ZW4gcHJvZmVzc2lvbmFsIGRvbWFpbnMuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBSZXBvcnRlZCBjb21iaW5lZCBzZXR1cDogT3B1cyA1LjUgdy8gZmFsbGJhY2tzLiBGYWxsYmFjayBiZWhhdmlvciBtdXN0IG5vdCBiZSBhdHRyaWJ1dGVkIHRvIHRoZSBiYXNlIG1vZGVsIGFsb25lLg","id":"launch-openai-gpt61-sol-launch-2390","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Combined fallback setup is not the named base model at one effort setting.","comparable":false,"configuration":"Reported reasoning effort max. Professional questions about complex PDFs across ten professional domains. OpenAI research environment or API; same named metric and evaluation chart. Reported combined setup: Opus 5.5 w/ fallbacks. Fallback behavior must not be attributed to the base model alone.","family":"gdp.pdf","locator":"Chart: GDP.pdf / Opus 5.5 w/ fallbacks / max","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $1.55233548; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":26.2,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"automationbench-1-0-6-percent","evidenceKind":"lab_self_report","harnessId":"automationbench-1-0-6-percent:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIEVuZC10by1lbmQgYnVzaW5lc3Mgd29ya2Zsb3dzIHVzaW5nIDQ3IHRvb2xzIGFjcm9zcyBzaXggYnVzaW5lc3MgZnVuY3Rpb25zLiBPcGVuQUkgcmVzZWFyY2ggZW52aXJvbm1lbnQgb3IgQVBJOyBzYW1lIG5hbWVkIG1ldHJpYyBhbmQgZXZhbHVhdGlvbiBjaGFydC4gUmVwb3J0ZWQgY29tYmluZWQgc2V0dXA6IEZhYmxlIDUuMSB3LyBPcHVzIDUgZmFsbGJhY2suIEZhbGxiYWNrIGJlaGF2aW9yIG11c3Qgbm90IGJlIGF0dHJpYnV0ZWQgdG8gdGhlIGJhc2UgbW9kZWwgYWxvbmUu","id":"launch-openai-gpt61-sol-launch-2391","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Combined fallback setup is not the named base model at one effort setting.","comparable":false,"configuration":"Reported reasoning effort max. End-to-end business workflows using 47 tools across six business functions. OpenAI research environment or API; same named metric and evaluation chart. Reported combined setup: Fable 5.1 w/ Opus 5 fallback. Fallback behavior must not be attributed to the base model alone.","family":"AutomationBench","locator":"Chart: AutomationBench / Fable 5.1 w/ Opus 5 fallback / max","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $2.45; source-specific cost does not establish consensus workload efficiency. The launch caption says fallback costs are omitted and fallbacks occurred on about 40% of tasks. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":31.4,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"automationbench-1-0-6-percent","evidenceKind":"lab_self_report","harnessId":"automationbench-1-0-6-percent:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBsb3cuIEVuZC10by1lbmQgYnVzaW5lc3Mgd29ya2Zsb3dzIHVzaW5nIDQ3IHRvb2xzIGFjcm9zcyBzaXggYnVzaW5lc3MgZnVuY3Rpb25zLiBPcGVuQUkgcmVzZWFyY2ggZW52aXJvbm1lbnQgb3IgQVBJOyBzYW1lIG5hbWVkIG1ldHJpYyBhbmQgZXZhbHVhdGlvbiBjaGFydC4gTm8gZmFsbGJhY2sgaXMgcmVwb3J0ZWQgZm9yIHRoaXMgT3BlbkFJIGNvbmZpZ3VyYXRpb24u","id":"launch-openai-gpt61-sol-launch-2392","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort low. End-to-end business workflows using 47 tools across six business functions. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"AutomationBench","locator":"Chart: AutomationBench / GPT-6 Astra / low","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $1.08; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":30.3,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"automationbench-1-0-6-percent","evidenceKind":"lab_self_report","harnessId":"automationbench-1-0-6-percent:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtZWRpdW0uIEVuZC10by1lbmQgYnVzaW5lc3Mgd29ya2Zsb3dzIHVzaW5nIDQ3IHRvb2xzIGFjcm9zcyBzaXggYnVzaW5lc3MgZnVuY3Rpb25zLiBPcGVuQUkgcmVzZWFyY2ggZW52aXJvbm1lbnQgb3IgQVBJOyBzYW1lIG5hbWVkIG1ldHJpYyBhbmQgZXZhbHVhdGlvbiBjaGFydC4gTm8gZmFsbGJhY2sgaXMgcmVwb3J0ZWQgZm9yIHRoaXMgT3BlbkFJIGNvbmZpZ3VyYXRpb24u","id":"launch-openai-gpt61-sol-launch-2393","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort medium. End-to-end business workflows using 47 tools across six business functions. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"AutomationBench","locator":"Chart: AutomationBench / GPT-6 Astra / medium","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $1.27; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":34.1,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"automationbench-1-0-6-percent","evidenceKind":"lab_self_report","harnessId":"automationbench-1-0-6-percent:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBoaWdoLiBFbmQtdG8tZW5kIGJ1c2luZXNzIHdvcmtmbG93cyB1c2luZyA0NyB0b29scyBhY3Jvc3Mgc2l4IGJ1c2luZXNzIGZ1bmN0aW9ucy4gT3BlbkFJIHJlc2VhcmNoIGVudmlyb25tZW50IG9yIEFQSTsgc2FtZSBuYW1lZCBtZXRyaWMgYW5kIGV2YWx1YXRpb24gY2hhcnQuIE5vIGZhbGxiYWNrIGlzIHJlcG9ydGVkIGZvciB0aGlzIE9wZW5BSSBjb25maWd1cmF0aW9uLg","id":"launch-openai-gpt61-sol-launch-2394","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort high. End-to-end business workflows using 47 tools across six business functions. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"AutomationBench","locator":"Chart: AutomationBench / GPT-6 Astra / high","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $1.44; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":37.1,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"automationbench-1-0-6-percent","evidenceKind":"lab_self_report","harnessId":"automationbench-1-0-6-percent:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCB4aGlnaC4gRW5kLXRvLWVuZCBidXNpbmVzcyB3b3JrZmxvd3MgdXNpbmcgNDcgdG9vbHMgYWNyb3NzIHNpeCBidXNpbmVzcyBmdW5jdGlvbnMuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","id":"launch-openai-gpt61-sol-launch-2395","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort xhigh. End-to-end business workflows using 47 tools across six business functions. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"AutomationBench","locator":"Chart: AutomationBench / GPT-6 Astra / xhigh","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $1.5; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":39,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"automationbench-1-0-6-percent","evidenceKind":"lab_self_report","harnessId":"automationbench-1-0-6-percent:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIEVuZC10by1lbmQgYnVzaW5lc3Mgd29ya2Zsb3dzIHVzaW5nIDQ3IHRvb2xzIGFjcm9zcyBzaXggYnVzaW5lc3MgZnVuY3Rpb25zLiBPcGVuQUkgcmVzZWFyY2ggZW52aXJvbm1lbnQgb3IgQVBJOyBzYW1lIG5hbWVkIG1ldHJpYyBhbmQgZXZhbHVhdGlvbiBjaGFydC4gTm8gZmFsbGJhY2sgaXMgcmVwb3J0ZWQgZm9yIHRoaXMgT3BlbkFJIGNvbmZpZ3VyYXRpb24u","id":"launch-openai-gpt61-sol-launch-2396","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort max. End-to-end business workflows using 47 tools across six business functions. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"AutomationBench","locator":"Chart: AutomationBench / GPT-6 Astra / max","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $1.73; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":41.4,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"automationbench-1-0-6-percent","evidenceKind":"lab_self_report","harnessId":"automationbench-1-0-6-percent:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBsb3cuIEVuZC10by1lbmQgYnVzaW5lc3Mgd29ya2Zsb3dzIHVzaW5nIDQ3IHRvb2xzIGFjcm9zcyBzaXggYnVzaW5lc3MgZnVuY3Rpb25zLiBPcGVuQUkgcmVzZWFyY2ggZW52aXJvbm1lbnQgb3IgQVBJOyBzYW1lIG5hbWVkIG1ldHJpYyBhbmQgZXZhbHVhdGlvbiBjaGFydC4gTm8gZmFsbGJhY2sgaXMgcmVwb3J0ZWQgZm9yIHRoaXMgT3BlbkFJIGNvbmZpZ3VyYXRpb24u","id":"launch-openai-gpt61-sol-launch-2397","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6.1-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort low. End-to-end business workflows using 47 tools across six business functions. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"AutomationBench","locator":"Chart: AutomationBench / GPT-6.1 Sol / low","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $0.157; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":24.7,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"automationbench-1-0-6-percent","evidenceKind":"lab_self_report","harnessId":"automationbench-1-0-6-percent:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtZWRpdW0uIEVuZC10by1lbmQgYnVzaW5lc3Mgd29ya2Zsb3dzIHVzaW5nIDQ3IHRvb2xzIGFjcm9zcyBzaXggYnVzaW5lc3MgZnVuY3Rpb25zLiBPcGVuQUkgcmVzZWFyY2ggZW52aXJvbm1lbnQgb3IgQVBJOyBzYW1lIG5hbWVkIG1ldHJpYyBhbmQgZXZhbHVhdGlvbiBjaGFydC4gTm8gZmFsbGJhY2sgaXMgcmVwb3J0ZWQgZm9yIHRoaXMgT3BlbkFJIGNvbmZpZ3VyYXRpb24u","id":"launch-openai-gpt61-sol-launch-2398","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6.1-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort medium. End-to-end business workflows using 47 tools across six business functions. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"AutomationBench","locator":"Chart: AutomationBench / GPT-6.1 Sol / medium","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $0.1917; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":31.7,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"automationbench-1-0-6-percent","evidenceKind":"lab_self_report","harnessId":"automationbench-1-0-6-percent:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBoaWdoLiBFbmQtdG8tZW5kIGJ1c2luZXNzIHdvcmtmbG93cyB1c2luZyA0NyB0b29scyBhY3Jvc3Mgc2l4IGJ1c2luZXNzIGZ1bmN0aW9ucy4gT3BlbkFJIHJlc2VhcmNoIGVudmlyb25tZW50IG9yIEFQSTsgc2FtZSBuYW1lZCBtZXRyaWMgYW5kIGV2YWx1YXRpb24gY2hhcnQuIE5vIGZhbGxiYWNrIGlzIHJlcG9ydGVkIGZvciB0aGlzIE9wZW5BSSBjb25maWd1cmF0aW9uLg","id":"launch-openai-gpt61-sol-launch-2399","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6.1-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort high. End-to-end business workflows using 47 tools across six business functions. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"AutomationBench","locator":"Chart: AutomationBench / GPT-6.1 Sol / high","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $0.2255; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":33.2,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"automationbench-1-0-6-percent","evidenceKind":"lab_self_report","harnessId":"automationbench-1-0-6-percent:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCB4aGlnaC4gRW5kLXRvLWVuZCBidXNpbmVzcyB3b3JrZmxvd3MgdXNpbmcgNDcgdG9vbHMgYWNyb3NzIHNpeCBidXNpbmVzcyBmdW5jdGlvbnMuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","id":"launch-openai-gpt61-sol-launch-2400","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6.1-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort xhigh. End-to-end business workflows using 47 tools across six business functions. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"AutomationBench","locator":"Chart: AutomationBench / GPT-6.1 Sol / xhigh","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $0.2508; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":35.5,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"automationbench-1-0-6-percent","evidenceKind":"lab_self_report","harnessId":"automationbench-1-0-6-percent:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIEVuZC10by1lbmQgYnVzaW5lc3Mgd29ya2Zsb3dzIHVzaW5nIDQ3IHRvb2xzIGFjcm9zcyBzaXggYnVzaW5lc3MgZnVuY3Rpb25zLiBPcGVuQUkgcmVzZWFyY2ggZW52aXJvbm1lbnQgb3IgQVBJOyBzYW1lIG5hbWVkIG1ldHJpYyBhbmQgZXZhbHVhdGlvbiBjaGFydC4gTm8gZmFsbGJhY2sgaXMgcmVwb3J0ZWQgZm9yIHRoaXMgT3BlbkFJIGNvbmZpZ3VyYXRpb24u","id":"launch-openai-gpt61-sol-launch-2401","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6.1-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort max. End-to-end business workflows using 47 tools across six business functions. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"AutomationBench","locator":"Chart: AutomationBench / GPT-6.1 Sol / max","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $0.2989; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":36.1,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"automationbench-1-0-6-percent","evidenceKind":"lab_self_report","harnessId":"automationbench-1-0-6-percent:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBsb3cuIEVuZC10by1lbmQgYnVzaW5lc3Mgd29ya2Zsb3dzIHVzaW5nIDQ3IHRvb2xzIGFjcm9zcyBzaXggYnVzaW5lc3MgZnVuY3Rpb25zLiBPcGVuQUkgcmVzZWFyY2ggZW52aXJvbm1lbnQgb3IgQVBJOyBzYW1lIG5hbWVkIG1ldHJpYyBhbmQgZXZhbHVhdGlvbiBjaGFydC4gTm8gZmFsbGJhY2sgaXMgcmVwb3J0ZWQgZm9yIHRoaXMgT3BlbkFJIGNvbmZpZ3VyYXRpb24u","id":"launch-openai-gpt61-sol-launch-2402","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort low. End-to-end business workflows using 47 tools across six business functions. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"AutomationBench","locator":"Chart: AutomationBench / GPT-6 Sol / low","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $0.1861; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":21.16,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"automationbench-1-0-6-percent","evidenceKind":"lab_self_report","harnessId":"automationbench-1-0-6-percent:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtZWRpdW0uIEVuZC10by1lbmQgYnVzaW5lc3Mgd29ya2Zsb3dzIHVzaW5nIDQ3IHRvb2xzIGFjcm9zcyBzaXggYnVzaW5lc3MgZnVuY3Rpb25zLiBPcGVuQUkgcmVzZWFyY2ggZW52aXJvbm1lbnQgb3IgQVBJOyBzYW1lIG5hbWVkIG1ldHJpYyBhbmQgZXZhbHVhdGlvbiBjaGFydC4gTm8gZmFsbGJhY2sgaXMgcmVwb3J0ZWQgZm9yIHRoaXMgT3BlbkFJIGNvbmZpZ3VyYXRpb24u","id":"launch-openai-gpt61-sol-launch-2403","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort medium. End-to-end business workflows using 47 tools across six business functions. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"AutomationBench","locator":"Chart: AutomationBench / GPT-6 Sol / medium","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $0.2098; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":26.94,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"automationbench-1-0-6-percent","evidenceKind":"lab_self_report","harnessId":"automationbench-1-0-6-percent:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBoaWdoLiBFbmQtdG8tZW5kIGJ1c2luZXNzIHdvcmtmbG93cyB1c2luZyA0NyB0b29scyBhY3Jvc3Mgc2l4IGJ1c2luZXNzIGZ1bmN0aW9ucy4gT3BlbkFJIHJlc2VhcmNoIGVudmlyb25tZW50IG9yIEFQSTsgc2FtZSBuYW1lZCBtZXRyaWMgYW5kIGV2YWx1YXRpb24gY2hhcnQuIE5vIGZhbGxiYWNrIGlzIHJlcG9ydGVkIGZvciB0aGlzIE9wZW5BSSBjb25maWd1cmF0aW9uLg","id":"launch-openai-gpt61-sol-launch-2404","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort high. End-to-end business workflows using 47 tools across six business functions. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"AutomationBench","locator":"Chart: AutomationBench / GPT-6 Sol / high","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $0.2368; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":31.2,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"automationbench-1-0-6-percent","evidenceKind":"lab_self_report","harnessId":"automationbench-1-0-6-percent:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCB4aGlnaC4gRW5kLXRvLWVuZCBidXNpbmVzcyB3b3JrZmxvd3MgdXNpbmcgNDcgdG9vbHMgYWNyb3NzIHNpeCBidXNpbmVzcyBmdW5jdGlvbnMuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","id":"launch-openai-gpt61-sol-launch-2405","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort xhigh. End-to-end business workflows using 47 tools across six business functions. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"AutomationBench","locator":"Chart: AutomationBench / GPT-6 Sol / xhigh","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $0.2746; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":33.18,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"automationbench-1-0-6-percent","evidenceKind":"lab_self_report","harnessId":"automationbench-1-0-6-percent:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIEVuZC10by1lbmQgYnVzaW5lc3Mgd29ya2Zsb3dzIHVzaW5nIDQ3IHRvb2xzIGFjcm9zcyBzaXggYnVzaW5lc3MgZnVuY3Rpb25zLiBPcGVuQUkgcmVzZWFyY2ggZW52aXJvbm1lbnQgb3IgQVBJOyBzYW1lIG5hbWVkIG1ldHJpYyBhbmQgZXZhbHVhdGlvbiBjaGFydC4gTm8gZmFsbGJhY2sgaXMgcmVwb3J0ZWQgZm9yIHRoaXMgT3BlbkFJIGNvbmZpZ3VyYXRpb24u","id":"launch-openai-gpt61-sol-launch-2406","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort max. End-to-end business workflows using 47 tools across six business functions. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"AutomationBench","locator":"Chart: AutomationBench / GPT-6 Sol / max","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $0.3406; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":31.96,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"automationbench-1-0-6-percent","evidenceKind":"lab_self_report","harnessId":"automationbench-1-0-6-percent:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBsb3cuIEVuZC10by1lbmQgYnVzaW5lc3Mgd29ya2Zsb3dzIHVzaW5nIDQ3IHRvb2xzIGFjcm9zcyBzaXggYnVzaW5lc3MgZnVuY3Rpb25zLiBPcGVuQUkgcmVzZWFyY2ggZW52aXJvbm1lbnQgb3IgQVBJOyBzYW1lIG5hbWVkIG1ldHJpYyBhbmQgZXZhbHVhdGlvbiBjaGFydC4gUmVwb3J0ZWQgY29tYmluZWQgc2V0dXA6IE9wdXMgNS41IHcvIGZhbGxiYWNrcy4gRmFsbGJhY2sgYmVoYXZpb3IgbXVzdCBub3QgYmUgYXR0cmlidXRlZCB0byB0aGUgYmFzZSBtb2RlbCBhbG9uZS4","id":"launch-openai-gpt61-sol-launch-2407","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Combined fallback setup is not the named base model at one effort setting.","comparable":false,"configuration":"Reported reasoning effort low. End-to-end business workflows using 47 tools across six business functions. OpenAI research environment or API; same named metric and evaluation chart. Reported combined setup: Opus 5.5 w/ fallbacks. Fallback behavior must not be attributed to the base model alone.","family":"AutomationBench","locator":"Chart: AutomationBench / Opus 5.5 w/ fallbacks / low","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $0.51; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":24.2,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"automationbench-1-0-6-percent","evidenceKind":"lab_self_report","harnessId":"automationbench-1-0-6-percent:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtZWRpdW0uIEVuZC10by1lbmQgYnVzaW5lc3Mgd29ya2Zsb3dzIHVzaW5nIDQ3IHRvb2xzIGFjcm9zcyBzaXggYnVzaW5lc3MgZnVuY3Rpb25zLiBPcGVuQUkgcmVzZWFyY2ggZW52aXJvbm1lbnQgb3IgQVBJOyBzYW1lIG5hbWVkIG1ldHJpYyBhbmQgZXZhbHVhdGlvbiBjaGFydC4gUmVwb3J0ZWQgY29tYmluZWQgc2V0dXA6IE9wdXMgNS41IHcvIGZhbGxiYWNrcy4gRmFsbGJhY2sgYmVoYXZpb3IgbXVzdCBub3QgYmUgYXR0cmlidXRlZCB0byB0aGUgYmFzZSBtb2RlbCBhbG9uZS4","id":"launch-openai-gpt61-sol-launch-2408","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Combined fallback setup is not the named base model at one effort setting.","comparable":false,"configuration":"Reported reasoning effort medium. End-to-end business workflows using 47 tools across six business functions. OpenAI research environment or API; same named metric and evaluation chart. Reported combined setup: Opus 5.5 w/ fallbacks. Fallback behavior must not be attributed to the base model alone.","family":"AutomationBench","locator":"Chart: AutomationBench / Opus 5.5 w/ fallbacks / medium","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $0.65; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":29.53,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"automationbench-1-0-6-percent","evidenceKind":"lab_self_report","harnessId":"automationbench-1-0-6-percent:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBoaWdoLiBFbmQtdG8tZW5kIGJ1c2luZXNzIHdvcmtmbG93cyB1c2luZyA0NyB0b29scyBhY3Jvc3Mgc2l4IGJ1c2luZXNzIGZ1bmN0aW9ucy4gT3BlbkFJIHJlc2VhcmNoIGVudmlyb25tZW50IG9yIEFQSTsgc2FtZSBuYW1lZCBtZXRyaWMgYW5kIGV2YWx1YXRpb24gY2hhcnQuIFJlcG9ydGVkIGNvbWJpbmVkIHNldHVwOiBPcHVzIDUuNSB3LyBmYWxsYmFja3MuIEZhbGxiYWNrIGJlaGF2aW9yIG11c3Qgbm90IGJlIGF0dHJpYnV0ZWQgdG8gdGhlIGJhc2UgbW9kZWwgYWxvbmUu","id":"launch-openai-gpt61-sol-launch-2409","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Combined fallback setup is not the named base model at one effort setting.","comparable":false,"configuration":"Reported reasoning effort high. End-to-end business workflows using 47 tools across six business functions. OpenAI research environment or API; same named metric and evaluation chart. Reported combined setup: Opus 5.5 w/ fallbacks. Fallback behavior must not be attributed to the base model alone.","family":"AutomationBench","locator":"Chart: AutomationBench / Opus 5.5 w/ fallbacks / high","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $0.71; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":33.03,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"automationbench-1-0-6-percent","evidenceKind":"lab_self_report","harnessId":"automationbench-1-0-6-percent:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCB4aGlnaC4gRW5kLXRvLWVuZCBidXNpbmVzcyB3b3JrZmxvd3MgdXNpbmcgNDcgdG9vbHMgYWNyb3NzIHNpeCBidXNpbmVzcyBmdW5jdGlvbnMuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBSZXBvcnRlZCBjb21iaW5lZCBzZXR1cDogT3B1cyA1LjUgdy8gZmFsbGJhY2tzLiBGYWxsYmFjayBiZWhhdmlvciBtdXN0IG5vdCBiZSBhdHRyaWJ1dGVkIHRvIHRoZSBiYXNlIG1vZGVsIGFsb25lLg","id":"launch-openai-gpt61-sol-launch-2410","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Combined fallback setup is not the named base model at one effort setting.","comparable":false,"configuration":"Reported reasoning effort xhigh. End-to-end business workflows using 47 tools across six business functions. OpenAI research environment or API; same named metric and evaluation chart. Reported combined setup: Opus 5.5 w/ fallbacks. Fallback behavior must not be attributed to the base model alone.","family":"AutomationBench","locator":"Chart: AutomationBench / Opus 5.5 w/ fallbacks / xhigh","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $0.89; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":35.77,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"automationbench-1-0-6-percent","evidenceKind":"lab_self_report","harnessId":"automationbench-1-0-6-percent:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIEVuZC10by1lbmQgYnVzaW5lc3Mgd29ya2Zsb3dzIHVzaW5nIDQ3IHRvb2xzIGFjcm9zcyBzaXggYnVzaW5lc3MgZnVuY3Rpb25zLiBPcGVuQUkgcmVzZWFyY2ggZW52aXJvbm1lbnQgb3IgQVBJOyBzYW1lIG5hbWVkIG1ldHJpYyBhbmQgZXZhbHVhdGlvbiBjaGFydC4gUmVwb3J0ZWQgY29tYmluZWQgc2V0dXA6IE9wdXMgNS41IHcvIGZhbGxiYWNrcy4gRmFsbGJhY2sgYmVoYXZpb3IgbXVzdCBub3QgYmUgYXR0cmlidXRlZCB0byB0aGUgYmFzZSBtb2RlbCBhbG9uZS4","id":"launch-openai-gpt61-sol-launch-2411","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Combined fallback setup is not the named base model at one effort setting.","comparable":false,"configuration":"Reported reasoning effort max. End-to-end business workflows using 47 tools across six business functions. OpenAI research environment or API; same named metric and evaluation chart. Reported combined setup: Opus 5.5 w/ fallbacks. Fallback behavior must not be attributed to the base model alone.","family":"AutomationBench","locator":"Chart: AutomationBench / Opus 5.5 w/ fallbacks / max","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $1.44; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":42.47,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline","evidenceKind":"lab_self_report","harnessId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBsb3cuIE9mZmxpbmUgc2V0IGZyb20gdjIwMjYuMDguMDg7IHBhcnRpYWwgcmV3YXJkLCBub3QgYmluYXJ5IGZ1bGwtdGFzayBzdWNjZXNzIG9yIE9TV29ybGQgVmVyaWZpZWQuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","id":"launch-openai-gpt61-sol-launch-2412","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort low. Offline set from v2026.08.08; partial reward, not binary full-task success or OSWorld Verified. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"OSWorld","locator":"Chart: OSWorld 2.0, offline set / GPT-6 Astra / low","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $2.7159; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":62.17,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline","evidenceKind":"lab_self_report","harnessId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtZWRpdW0uIE9mZmxpbmUgc2V0IGZyb20gdjIwMjYuMDguMDg7IHBhcnRpYWwgcmV3YXJkLCBub3QgYmluYXJ5IGZ1bGwtdGFzayBzdWNjZXNzIG9yIE9TV29ybGQgVmVyaWZpZWQuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","id":"launch-openai-gpt61-sol-launch-2413","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort medium. Offline set from v2026.08.08; partial reward, not binary full-task success or OSWorld Verified. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"OSWorld","locator":"Chart: OSWorld 2.0, offline set / GPT-6 Astra / medium","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $5.3594; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":69.25,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline","evidenceKind":"lab_self_report","harnessId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBoaWdoLiBPZmZsaW5lIHNldCBmcm9tIHYyMDI2LjA4LjA4OyBwYXJ0aWFsIHJld2FyZCwgbm90IGJpbmFyeSBmdWxsLXRhc2sgc3VjY2VzcyBvciBPU1dvcmxkIFZlcmlmaWVkLiBPcGVuQUkgcmVzZWFyY2ggZW52aXJvbm1lbnQgb3IgQVBJOyBzYW1lIG5hbWVkIG1ldHJpYyBhbmQgZXZhbHVhdGlvbiBjaGFydC4gTm8gZmFsbGJhY2sgaXMgcmVwb3J0ZWQgZm9yIHRoaXMgT3BlbkFJIGNvbmZpZ3VyYXRpb24u","id":"launch-openai-gpt61-sol-launch-2414","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort high. Offline set from v2026.08.08; partial reward, not binary full-task success or OSWorld Verified. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"OSWorld","locator":"Chart: OSWorld 2.0, offline set / GPT-6 Astra / high","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $6.9064; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":70.02,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline","evidenceKind":"lab_self_report","harnessId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCB4aGlnaC4gT2ZmbGluZSBzZXQgZnJvbSB2MjAyNi4wOC4wODsgcGFydGlhbCByZXdhcmQsIG5vdCBiaW5hcnkgZnVsbC10YXNrIHN1Y2Nlc3Mgb3IgT1NXb3JsZCBWZXJpZmllZC4gT3BlbkFJIHJlc2VhcmNoIGVudmlyb25tZW50IG9yIEFQSTsgc2FtZSBuYW1lZCBtZXRyaWMgYW5kIGV2YWx1YXRpb24gY2hhcnQuIE5vIGZhbGxiYWNrIGlzIHJlcG9ydGVkIGZvciB0aGlzIE9wZW5BSSBjb25maWd1cmF0aW9uLg","id":"launch-openai-gpt61-sol-launch-2415","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort xhigh. Offline set from v2026.08.08; partial reward, not binary full-task success or OSWorld Verified. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"OSWorld","locator":"Chart: OSWorld 2.0, offline set / GPT-6 Astra / xhigh","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $7.4944; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":71.27,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline","evidenceKind":"lab_self_report","harnessId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIE9mZmxpbmUgc2V0IGZyb20gdjIwMjYuMDguMDg7IHBhcnRpYWwgcmV3YXJkLCBub3QgYmluYXJ5IGZ1bGwtdGFzayBzdWNjZXNzIG9yIE9TV29ybGQgVmVyaWZpZWQuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","id":"launch-openai-gpt61-sol-launch-2416","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort max. Offline set from v2026.08.08; partial reward, not binary full-task success or OSWorld Verified. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"OSWorld","locator":"Chart: OSWorld 2.0, offline set / GPT-6 Astra / max","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $9.4353; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":73.49,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline","evidenceKind":"lab_self_report","harnessId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBsb3cuIE9mZmxpbmUgc2V0IGZyb20gdjIwMjYuMDguMDg7IHBhcnRpYWwgcmV3YXJkLCBub3QgYmluYXJ5IGZ1bGwtdGFzayBzdWNjZXNzIG9yIE9TV29ybGQgVmVyaWZpZWQuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","id":"launch-openai-gpt61-sol-launch-2417","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6.1-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort low. Offline set from v2026.08.08; partial reward, not binary full-task success or OSWorld Verified. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"OSWorld","locator":"Chart: OSWorld 2.0, offline set / GPT-6.1 Sol / low","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $0.4248; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":58.96,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline","evidenceKind":"lab_self_report","harnessId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtZWRpdW0uIE9mZmxpbmUgc2V0IGZyb20gdjIwMjYuMDguMDg7IHBhcnRpYWwgcmV3YXJkLCBub3QgYmluYXJ5IGZ1bGwtdGFzayBzdWNjZXNzIG9yIE9TV29ybGQgVmVyaWZpZWQuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","id":"launch-openai-gpt61-sol-launch-2418","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6.1-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort medium. Offline set from v2026.08.08; partial reward, not binary full-task success or OSWorld Verified. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"OSWorld","locator":"Chart: OSWorld 2.0, offline set / GPT-6.1 Sol / medium","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $0.7675; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":66.84,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline","evidenceKind":"lab_self_report","harnessId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBoaWdoLiBPZmZsaW5lIHNldCBmcm9tIHYyMDI2LjA4LjA4OyBwYXJ0aWFsIHJld2FyZCwgbm90IGJpbmFyeSBmdWxsLXRhc2sgc3VjY2VzcyBvciBPU1dvcmxkIFZlcmlmaWVkLiBPcGVuQUkgcmVzZWFyY2ggZW52aXJvbm1lbnQgb3IgQVBJOyBzYW1lIG5hbWVkIG1ldHJpYyBhbmQgZXZhbHVhdGlvbiBjaGFydC4gTm8gZmFsbGJhY2sgaXMgcmVwb3J0ZWQgZm9yIHRoaXMgT3BlbkFJIGNvbmZpZ3VyYXRpb24u","id":"launch-openai-gpt61-sol-launch-2419","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6.1-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort high. Offline set from v2026.08.08; partial reward, not binary full-task success or OSWorld Verified. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"OSWorld","locator":"Chart: OSWorld 2.0, offline set / GPT-6.1 Sol / high","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $0.9613; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":69.56,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline","evidenceKind":"lab_self_report","harnessId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCB4aGlnaC4gT2ZmbGluZSBzZXQgZnJvbSB2MjAyNi4wOC4wODsgcGFydGlhbCByZXdhcmQsIG5vdCBiaW5hcnkgZnVsbC10YXNrIHN1Y2Nlc3Mgb3IgT1NXb3JsZCBWZXJpZmllZC4gT3BlbkFJIHJlc2VhcmNoIGVudmlyb25tZW50IG9yIEFQSTsgc2FtZSBuYW1lZCBtZXRyaWMgYW5kIGV2YWx1YXRpb24gY2hhcnQuIE5vIGZhbGxiYWNrIGlzIHJlcG9ydGVkIGZvciB0aGlzIE9wZW5BSSBjb25maWd1cmF0aW9uLg","id":"launch-openai-gpt61-sol-launch-2420","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6.1-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort xhigh. Offline set from v2026.08.08; partial reward, not binary full-task success or OSWorld Verified. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"OSWorld","locator":"Chart: OSWorld 2.0, offline set / GPT-6.1 Sol / xhigh","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $1.0466; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":69.38,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline","evidenceKind":"lab_self_report","harnessId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIE9mZmxpbmUgc2V0IGZyb20gdjIwMjYuMDguMDg7IHBhcnRpYWwgcmV3YXJkLCBub3QgYmluYXJ5IGZ1bGwtdGFzayBzdWNjZXNzIG9yIE9TV29ybGQgVmVyaWZpZWQuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","id":"launch-openai-gpt61-sol-launch-2421","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6.1-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort max. Offline set from v2026.08.08; partial reward, not binary full-task success or OSWorld Verified. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"OSWorld","locator":"Chart: OSWorld 2.0, offline set / GPT-6.1 Sol / max","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $1.2689; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":71.42,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline","evidenceKind":"lab_self_report","harnessId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBsb3cuIE9mZmxpbmUgc2V0IGZyb20gdjIwMjYuMDguMDg7IHBhcnRpYWwgcmV3YXJkLCBub3QgYmluYXJ5IGZ1bGwtdGFzayBzdWNjZXNzIG9yIE9TV29ybGQgVmVyaWZpZWQuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","id":"launch-openai-gpt61-sol-launch-2422","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort low. Offline set from v2026.08.08; partial reward, not binary full-task success or OSWorld Verified. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"OSWorld","locator":"Chart: OSWorld 2.0, offline set / GPT-6 Sol / low","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $1.0117; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":43.9,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline","evidenceKind":"lab_self_report","harnessId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtZWRpdW0uIE9mZmxpbmUgc2V0IGZyb20gdjIwMjYuMDguMDg7IHBhcnRpYWwgcmV3YXJkLCBub3QgYmluYXJ5IGZ1bGwtdGFzayBzdWNjZXNzIG9yIE9TV29ybGQgVmVyaWZpZWQuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","id":"launch-openai-gpt61-sol-launch-2423","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort medium. Offline set from v2026.08.08; partial reward, not binary full-task success or OSWorld Verified. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"OSWorld","locator":"Chart: OSWorld 2.0, offline set / GPT-6 Sol / medium","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $1.3789; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":54,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline","evidenceKind":"lab_self_report","harnessId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBoaWdoLiBPZmZsaW5lIHNldCBmcm9tIHYyMDI2LjA4LjA4OyBwYXJ0aWFsIHJld2FyZCwgbm90IGJpbmFyeSBmdWxsLXRhc2sgc3VjY2VzcyBvciBPU1dvcmxkIFZlcmlmaWVkLiBPcGVuQUkgcmVzZWFyY2ggZW52aXJvbm1lbnQgb3IgQVBJOyBzYW1lIG5hbWVkIG1ldHJpYyBhbmQgZXZhbHVhdGlvbiBjaGFydC4gTm8gZmFsbGJhY2sgaXMgcmVwb3J0ZWQgZm9yIHRoaXMgT3BlbkFJIGNvbmZpZ3VyYXRpb24u","id":"launch-openai-gpt61-sol-launch-2424","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort high. Offline set from v2026.08.08; partial reward, not binary full-task success or OSWorld Verified. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"OSWorld","locator":"Chart: OSWorld 2.0, offline set / GPT-6 Sol / high","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $1.7096; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":58.29,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline","evidenceKind":"lab_self_report","harnessId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCB4aGlnaC4gT2ZmbGluZSBzZXQgZnJvbSB2MjAyNi4wOC4wODsgcGFydGlhbCByZXdhcmQsIG5vdCBiaW5hcnkgZnVsbC10YXNrIHN1Y2Nlc3Mgb3IgT1NXb3JsZCBWZXJpZmllZC4gT3BlbkFJIHJlc2VhcmNoIGVudmlyb25tZW50IG9yIEFQSTsgc2FtZSBuYW1lZCBtZXRyaWMgYW5kIGV2YWx1YXRpb24gY2hhcnQuIE5vIGZhbGxiYWNrIGlzIHJlcG9ydGVkIGZvciB0aGlzIE9wZW5BSSBjb25maWd1cmF0aW9uLg","id":"launch-openai-gpt61-sol-launch-2425","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort xhigh. Offline set from v2026.08.08; partial reward, not binary full-task success or OSWorld Verified. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"OSWorld","locator":"Chart: OSWorld 2.0, offline set / GPT-6 Sol / xhigh","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $2.3002; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":60.54,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline","evidenceKind":"lab_self_report","harnessId":"osworld-2-0-v2026-08-08-offline-set-partial-score-2-0-v2026-08-08-offline:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIE9mZmxpbmUgc2V0IGZyb20gdjIwMjYuMDguMDg7IHBhcnRpYWwgcmV3YXJkLCBub3QgYmluYXJ5IGZ1bGwtdGFzayBzdWNjZXNzIG9yIE9TV29ybGQgVmVyaWZpZWQuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","id":"launch-openai-gpt61-sol-launch-2426","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort max. Offline set from v2026.08.08; partial reward, not binary full-task success or OSWorld Verified. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"OSWorld","locator":"Chart: OSWorld 2.0, offline set / GPT-6 Sol / max","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $3.3695; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":64.43,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"terminal-bench-science-0-1","evidenceKind":"lab_self_report","harnessId":"terminal-bench-science-0-1:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIFNjaWVudGlmaWMgcmVzZWFyY2ggd29ya2Zsb3dzIHVzaW5nIGNvZGUgYW5kIHRlcm1pbmFsIHRvb2xzLCBpbmNsdWRpbmcgYW5hbHlzaXMsIHNpbXVsYXRpb24gYW5kIG1vZGVsIGZpdHRpbmcuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBSZXBvcnRlZCBjb21iaW5lZCBzZXR1cDogT3B1cyA1LjUgdy8gZmFsbGJhY2tzLiBGYWxsYmFjayBiZWhhdmlvciBtdXN0IG5vdCBiZSBhdHRyaWJ1dGVkIHRvIHRoZSBiYXNlIG1vZGVsIGFsb25lLg","id":"launch-openai-gpt61-sol-launch-2427","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"claude-opus-5-5","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Combined fallback setup is not the named base model at one effort setting.","comparable":false,"configuration":"Reported reasoning effort max. Scientific research workflows using code and terminal tools, including analysis, simulation and model fitting. OpenAI research environment or API; same named metric and evaluation chart. Reported combined setup: Opus 5.5 w/ fallbacks. Fallback behavior must not be attributed to the base model alone.","family":"Terminal-Bench Science","locator":"Chart: Terminal-Bench Science 0.1 / Opus 5.5 w/ fallbacks / max","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $23.2108; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":63.33,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"terminal-bench-science-0-1","evidenceKind":"lab_self_report","harnessId":"terminal-bench-science-0-1:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBsb3cuIFNjaWVudGlmaWMgcmVzZWFyY2ggd29ya2Zsb3dzIHVzaW5nIGNvZGUgYW5kIHRlcm1pbmFsIHRvb2xzLCBpbmNsdWRpbmcgYW5hbHlzaXMsIHNpbXVsYXRpb24gYW5kIG1vZGVsIGZpdHRpbmcuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","id":"launch-openai-gpt61-sol-launch-2428","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort low. Scientific research workflows using code and terminal tools, including analysis, simulation and model fitting. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"Terminal-Bench Science","locator":"Chart: Terminal-Bench Science 0.1 / GPT-6 Astra / low","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $11.4061; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":55.43,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"terminal-bench-science-0-1","evidenceKind":"lab_self_report","harnessId":"terminal-bench-science-0-1:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtZWRpdW0uIFNjaWVudGlmaWMgcmVzZWFyY2ggd29ya2Zsb3dzIHVzaW5nIGNvZGUgYW5kIHRlcm1pbmFsIHRvb2xzLCBpbmNsdWRpbmcgYW5hbHlzaXMsIHNpbXVsYXRpb24gYW5kIG1vZGVsIGZpdHRpbmcuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","id":"launch-openai-gpt61-sol-launch-2429","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort medium. Scientific research workflows using code and terminal tools, including analysis, simulation and model fitting. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"Terminal-Bench Science","locator":"Chart: Terminal-Bench Science 0.1 / GPT-6 Astra / medium","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $12.3383; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":57.43,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"terminal-bench-science-0-1","evidenceKind":"lab_self_report","harnessId":"terminal-bench-science-0-1:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBoaWdoLiBTY2llbnRpZmljIHJlc2VhcmNoIHdvcmtmbG93cyB1c2luZyBjb2RlIGFuZCB0ZXJtaW5hbCB0b29scywgaW5jbHVkaW5nIGFuYWx5c2lzLCBzaW11bGF0aW9uIGFuZCBtb2RlbCBmaXR0aW5nLiBPcGVuQUkgcmVzZWFyY2ggZW52aXJvbm1lbnQgb3IgQVBJOyBzYW1lIG5hbWVkIG1ldHJpYyBhbmQgZXZhbHVhdGlvbiBjaGFydC4gTm8gZmFsbGJhY2sgaXMgcmVwb3J0ZWQgZm9yIHRoaXMgT3BlbkFJIGNvbmZpZ3VyYXRpb24u","id":"launch-openai-gpt61-sol-launch-2430","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort high. Scientific research workflows using code and terminal tools, including analysis, simulation and model fitting. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"Terminal-Bench Science","locator":"Chart: Terminal-Bench Science 0.1 / GPT-6 Astra / high","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $14.9534; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":62,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"terminal-bench-science-0-1","evidenceKind":"lab_self_report","harnessId":"terminal-bench-science-0-1:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCB4aGlnaC4gU2NpZW50aWZpYyByZXNlYXJjaCB3b3JrZmxvd3MgdXNpbmcgY29kZSBhbmQgdGVybWluYWwgdG9vbHMsIGluY2x1ZGluZyBhbmFseXNpcywgc2ltdWxhdGlvbiBhbmQgbW9kZWwgZml0dGluZy4gT3BlbkFJIHJlc2VhcmNoIGVudmlyb25tZW50IG9yIEFQSTsgc2FtZSBuYW1lZCBtZXRyaWMgYW5kIGV2YWx1YXRpb24gY2hhcnQuIE5vIGZhbGxiYWNrIGlzIHJlcG9ydGVkIGZvciB0aGlzIE9wZW5BSSBjb25maWd1cmF0aW9uLg","id":"launch-openai-gpt61-sol-launch-2431","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort xhigh. Scientific research workflows using code and terminal tools, including analysis, simulation and model fitting. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"Terminal-Bench Science","locator":"Chart: Terminal-Bench Science 0.1 / GPT-6 Astra / xhigh","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $15.7554; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":60.86,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"terminal-bench-science-0-1","evidenceKind":"lab_self_report","harnessId":"terminal-bench-science-0-1:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIFNjaWVudGlmaWMgcmVzZWFyY2ggd29ya2Zsb3dzIHVzaW5nIGNvZGUgYW5kIHRlcm1pbmFsIHRvb2xzLCBpbmNsdWRpbmcgYW5hbHlzaXMsIHNpbXVsYXRpb24gYW5kIG1vZGVsIGZpdHRpbmcuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","id":"launch-openai-gpt61-sol-launch-2432","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort max. Scientific research workflows using code and terminal tools, including analysis, simulation and model fitting. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"Terminal-Bench Science","locator":"Chart: Terminal-Bench Science 0.1 / GPT-6 Astra / max","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $23.7974; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":68.1,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"terminal-bench-science-0-1","evidenceKind":"lab_self_report","harnessId":"terminal-bench-science-0-1:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBsb3cuIFNjaWVudGlmaWMgcmVzZWFyY2ggd29ya2Zsb3dzIHVzaW5nIGNvZGUgYW5kIHRlcm1pbmFsIHRvb2xzLCBpbmNsdWRpbmcgYW5hbHlzaXMsIHNpbXVsYXRpb24gYW5kIG1vZGVsIGZpdHRpbmcuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","id":"launch-openai-gpt61-sol-launch-2433","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6.1-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort low. Scientific research workflows using code and terminal tools, including analysis, simulation and model fitting. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"Terminal-Bench Science","locator":"Chart: Terminal-Bench Science 0.1 / GPT-6.1 Sol / low","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $1.7918; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":43.71,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"terminal-bench-science-0-1","evidenceKind":"lab_self_report","harnessId":"terminal-bench-science-0-1:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtZWRpdW0uIFNjaWVudGlmaWMgcmVzZWFyY2ggd29ya2Zsb3dzIHVzaW5nIGNvZGUgYW5kIHRlcm1pbmFsIHRvb2xzLCBpbmNsdWRpbmcgYW5hbHlzaXMsIHNpbXVsYXRpb24gYW5kIG1vZGVsIGZpdHRpbmcuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","id":"launch-openai-gpt61-sol-launch-2434","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6.1-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort medium. Scientific research workflows using code and terminal tools, including analysis, simulation and model fitting. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"Terminal-Bench Science","locator":"Chart: Terminal-Bench Science 0.1 / GPT-6.1 Sol / medium","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $2.3386; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":47.56,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"terminal-bench-science-0-1","evidenceKind":"lab_self_report","harnessId":"terminal-bench-science-0-1:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBoaWdoLiBTY2llbnRpZmljIHJlc2VhcmNoIHdvcmtmbG93cyB1c2luZyBjb2RlIGFuZCB0ZXJtaW5hbCB0b29scywgaW5jbHVkaW5nIGFuYWx5c2lzLCBzaW11bGF0aW9uIGFuZCBtb2RlbCBmaXR0aW5nLiBPcGVuQUkgcmVzZWFyY2ggZW52aXJvbm1lbnQgb3IgQVBJOyBzYW1lIG5hbWVkIG1ldHJpYyBhbmQgZXZhbHVhdGlvbiBjaGFydC4gTm8gZmFsbGJhY2sgaXMgcmVwb3J0ZWQgZm9yIHRoaXMgT3BlbkFJIGNvbmZpZ3VyYXRpb24u","id":"launch-openai-gpt61-sol-launch-2435","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6.1-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort high. Scientific research workflows using code and terminal tools, including analysis, simulation and model fitting. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"Terminal-Bench Science","locator":"Chart: Terminal-Bench Science 0.1 / GPT-6.1 Sol / high","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $2.7594; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":51.14,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"terminal-bench-science-0-1","evidenceKind":"lab_self_report","harnessId":"terminal-bench-science-0-1:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCB4aGlnaC4gU2NpZW50aWZpYyByZXNlYXJjaCB3b3JrZmxvd3MgdXNpbmcgY29kZSBhbmQgdGVybWluYWwgdG9vbHMsIGluY2x1ZGluZyBhbmFseXNpcywgc2ltdWxhdGlvbiBhbmQgbW9kZWwgZml0dGluZy4gT3BlbkFJIHJlc2VhcmNoIGVudmlyb25tZW50IG9yIEFQSTsgc2FtZSBuYW1lZCBtZXRyaWMgYW5kIGV2YWx1YXRpb24gY2hhcnQuIE5vIGZhbGxiYWNrIGlzIHJlcG9ydGVkIGZvciB0aGlzIE9wZW5BSSBjb25maWd1cmF0aW9uLg","id":"launch-openai-gpt61-sol-launch-2436","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6.1-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort xhigh. Scientific research workflows using code and terminal tools, including analysis, simulation and model fitting. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"Terminal-Bench Science","locator":"Chart: Terminal-Bench Science 0.1 / GPT-6.1 Sol / xhigh","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $2.8922; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":53.71,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"terminal-bench-science-0-1","evidenceKind":"lab_self_report","harnessId":"terminal-bench-science-0-1:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIFNjaWVudGlmaWMgcmVzZWFyY2ggd29ya2Zsb3dzIHVzaW5nIGNvZGUgYW5kIHRlcm1pbmFsIHRvb2xzLCBpbmNsdWRpbmcgYW5hbHlzaXMsIHNpbXVsYXRpb24gYW5kIG1vZGVsIGZpdHRpbmcuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","id":"launch-openai-gpt61-sol-launch-2437","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6.1-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort max. Scientific research workflows using code and terminal tools, including analysis, simulation and model fitting. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"Terminal-Bench Science","locator":"Chart: Terminal-Bench Science 0.1 / GPT-6.1 Sol / max","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $5.4652; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":57.02,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"terminal-bench-science-0-1","evidenceKind":"lab_self_report","harnessId":"terminal-bench-science-0-1:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBsb3cuIFNjaWVudGlmaWMgcmVzZWFyY2ggd29ya2Zsb3dzIHVzaW5nIGNvZGUgYW5kIHRlcm1pbmFsIHRvb2xzLCBpbmNsdWRpbmcgYW5hbHlzaXMsIHNpbXVsYXRpb24gYW5kIG1vZGVsIGZpdHRpbmcuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","id":"launch-openai-gpt61-sol-launch-2438","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort low. Scientific research workflows using code and terminal tools, including analysis, simulation and model fitting. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"Terminal-Bench Science","locator":"Chart: Terminal-Bench Science 0.1 / GPT-6 Sol / low","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $2.9986; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":9.17,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"terminal-bench-science-0-1","evidenceKind":"lab_self_report","harnessId":"terminal-bench-science-0-1:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtZWRpdW0uIFNjaWVudGlmaWMgcmVzZWFyY2ggd29ya2Zsb3dzIHVzaW5nIGNvZGUgYW5kIHRlcm1pbmFsIHRvb2xzLCBpbmNsdWRpbmcgYW5hbHlzaXMsIHNpbXVsYXRpb24gYW5kIG1vZGVsIGZpdHRpbmcuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","id":"launch-openai-gpt61-sol-launch-2439","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort medium. Scientific research workflows using code and terminal tools, including analysis, simulation and model fitting. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"Terminal-Bench Science","locator":"Chart: Terminal-Bench Science 0.1 / GPT-6 Sol / medium","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $4.4068; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":14.49,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"terminal-bench-science-0-1","evidenceKind":"lab_self_report","harnessId":"terminal-bench-science-0-1:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBoaWdoLiBTY2llbnRpZmljIHJlc2VhcmNoIHdvcmtmbG93cyB1c2luZyBjb2RlIGFuZCB0ZXJtaW5hbCB0b29scywgaW5jbHVkaW5nIGFuYWx5c2lzLCBzaW11bGF0aW9uIGFuZCBtb2RlbCBmaXR0aW5nLiBPcGVuQUkgcmVzZWFyY2ggZW52aXJvbm1lbnQgb3IgQVBJOyBzYW1lIG5hbWVkIG1ldHJpYyBhbmQgZXZhbHVhdGlvbiBjaGFydC4gTm8gZmFsbGJhY2sgaXMgcmVwb3J0ZWQgZm9yIHRoaXMgT3BlbkFJIGNvbmZpZ3VyYXRpb24u","id":"launch-openai-gpt61-sol-launch-2440","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort high. Scientific research workflows using code and terminal tools, including analysis, simulation and model fitting. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"Terminal-Bench Science","locator":"Chart: Terminal-Bench Science 0.1 / GPT-6 Sol / high","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $4.6325; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":14.61,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"terminal-bench-science-0-1","evidenceKind":"lab_self_report","harnessId":"terminal-bench-science-0-1:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCB4aGlnaC4gU2NpZW50aWZpYyByZXNlYXJjaCB3b3JrZmxvd3MgdXNpbmcgY29kZSBhbmQgdGVybWluYWwgdG9vbHMsIGluY2x1ZGluZyBhbmFseXNpcywgc2ltdWxhdGlvbiBhbmQgbW9kZWwgZml0dGluZy4gT3BlbkFJIHJlc2VhcmNoIGVudmlyb25tZW50IG9yIEFQSTsgc2FtZSBuYW1lZCBtZXRyaWMgYW5kIGV2YWx1YXRpb24gY2hhcnQuIE5vIGZhbGxiYWNrIGlzIHJlcG9ydGVkIGZvciB0aGlzIE9wZW5BSSBjb25maWd1cmF0aW9uLg","id":"launch-openai-gpt61-sol-launch-2441","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort xhigh. Scientific research workflows using code and terminal tools, including analysis, simulation and model fitting. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"Terminal-Bench Science","locator":"Chart: Terminal-Bench Science 0.1 / GPT-6 Sol / xhigh","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $6.7698; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":25.29,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"terminal-bench-science-0-1","evidenceKind":"lab_self_report","harnessId":"terminal-bench-science-0-1:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIFNjaWVudGlmaWMgcmVzZWFyY2ggd29ya2Zsb3dzIHVzaW5nIGNvZGUgYW5kIHRlcm1pbmFsIHRvb2xzLCBpbmNsdWRpbmcgYW5hbHlzaXMsIHNpbXVsYXRpb24gYW5kIG1vZGVsIGZpdHRpbmcuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","id":"launch-openai-gpt61-sol-launch-2442","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Matched OpenAI configurations within this named chart. This does not establish equality with any independent evaluator harness or production ChatGPT behavior.","comparable":true,"configuration":"Reported reasoning effort max. Scientific research workflows using code and terminal tools, including analysis, simulation and model fitting. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"Terminal-Bench Science","locator":"Chart: Terminal-Bench Science 0.1 / GPT-6 Sol / max","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $12.1803; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":27.59,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"factual-error-rate-on-difficult-prompts-sol-6-1-launch-user-flagged-conversations","evidenceKind":"lab_self_report","harnessId":"factual-error-rate-on-difficult-prompts-sol-6-1-launch-user-flagged-conversations:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBsb3cuIFNoYXJlIG9mIGFuc3dlcnMgd2l0aCBhdCBsZWFzdCBvbmUgZmFjdHVhbCBlcnJvciBvbiBkZS1pZGVudGlmaWVkIGNvbnZlcnNhdGlvbnMgd2hlcmUgdXNlcnMgZmxhZ2dlZCBhbiBlYXJsaWVyIG1vZGVsIGVycm9yOyBkZWxpYmVyYXRlbHkgZGlmZmljdWx0LCBub3QgcmVwcmVzZW50YXRpdmUgb2YgdHlwaWNhbCB1c2FnZS4gT3BlbkFJIHJlc2VhcmNoIGVudmlyb25tZW50IG9yIEFQSTsgc2FtZSBuYW1lZCBtZXRyaWMgYW5kIGV2YWx1YXRpb24gY2hhcnQuIE5vIGZhbGxiYWNrIGlzIHJlcG9ydGVkIGZvciB0aGlzIE9wZW5BSSBjb25maWd1cmF0aW9uLg","id":"launch-openai-gpt61-sol-launch-2443","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"knowledge","comparabilityReason":"Private user-flagged prompt protocol needs separate review before capability scoring; retained as an exact lower-is-better sourced claim.","comparable":false,"configuration":"Reported reasoning effort low. Share of answers with at least one factual error on de-identified conversations where users flagged an earlier model error; deliberately difficult, not representative of typical usage. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"Factual error rate on difficult prompts","locator":"Chart: Factual error rate on difficult prompts (lower is better) / GPT-6 Astra / low","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $0.2417; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":6.26,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"factual-error-rate-on-difficult-prompts-sol-6-1-launch-user-flagged-conversations","evidenceKind":"lab_self_report","harnessId":"factual-error-rate-on-difficult-prompts-sol-6-1-launch-user-flagged-conversations:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtZWRpdW0uIFNoYXJlIG9mIGFuc3dlcnMgd2l0aCBhdCBsZWFzdCBvbmUgZmFjdHVhbCBlcnJvciBvbiBkZS1pZGVudGlmaWVkIGNvbnZlcnNhdGlvbnMgd2hlcmUgdXNlcnMgZmxhZ2dlZCBhbiBlYXJsaWVyIG1vZGVsIGVycm9yOyBkZWxpYmVyYXRlbHkgZGlmZmljdWx0LCBub3QgcmVwcmVzZW50YXRpdmUgb2YgdHlwaWNhbCB1c2FnZS4gT3BlbkFJIHJlc2VhcmNoIGVudmlyb25tZW50IG9yIEFQSTsgc2FtZSBuYW1lZCBtZXRyaWMgYW5kIGV2YWx1YXRpb24gY2hhcnQuIE5vIGZhbGxiYWNrIGlzIHJlcG9ydGVkIGZvciB0aGlzIE9wZW5BSSBjb25maWd1cmF0aW9uLg","id":"launch-openai-gpt61-sol-launch-2444","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"knowledge","comparabilityReason":"Private user-flagged prompt protocol needs separate review before capability scoring; retained as an exact lower-is-better sourced claim.","comparable":false,"configuration":"Reported reasoning effort medium. Share of answers with at least one factual error on de-identified conversations where users flagged an earlier model error; deliberately difficult, not representative of typical usage. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"Factual error rate on difficult prompts","locator":"Chart: Factual error rate on difficult prompts (lower is better) / GPT-6 Astra / medium","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $0.3103; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":4.41,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"factual-error-rate-on-difficult-prompts-sol-6-1-launch-user-flagged-conversations","evidenceKind":"lab_self_report","harnessId":"factual-error-rate-on-difficult-prompts-sol-6-1-launch-user-flagged-conversations:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBoaWdoLiBTaGFyZSBvZiBhbnN3ZXJzIHdpdGggYXQgbGVhc3Qgb25lIGZhY3R1YWwgZXJyb3Igb24gZGUtaWRlbnRpZmllZCBjb252ZXJzYXRpb25zIHdoZXJlIHVzZXJzIGZsYWdnZWQgYW4gZWFybGllciBtb2RlbCBlcnJvcjsgZGVsaWJlcmF0ZWx5IGRpZmZpY3VsdCwgbm90IHJlcHJlc2VudGF0aXZlIG9mIHR5cGljYWwgdXNhZ2UuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","id":"launch-openai-gpt61-sol-launch-2445","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"knowledge","comparabilityReason":"Private user-flagged prompt protocol needs separate review before capability scoring; retained as an exact lower-is-better sourced claim.","comparable":false,"configuration":"Reported reasoning effort high. Share of answers with at least one factual error on de-identified conversations where users flagged an earlier model error; deliberately difficult, not representative of typical usage. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"Factual error rate on difficult prompts","locator":"Chart: Factual error rate on difficult prompts (lower is better) / GPT-6 Astra / high","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $0.477; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":3.9,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"factual-error-rate-on-difficult-prompts-sol-6-1-launch-user-flagged-conversations","evidenceKind":"lab_self_report","harnessId":"factual-error-rate-on-difficult-prompts-sol-6-1-launch-user-flagged-conversations:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCB4aGlnaC4gU2hhcmUgb2YgYW5zd2VycyB3aXRoIGF0IGxlYXN0IG9uZSBmYWN0dWFsIGVycm9yIG9uIGRlLWlkZW50aWZpZWQgY29udmVyc2F0aW9ucyB3aGVyZSB1c2VycyBmbGFnZ2VkIGFuIGVhcmxpZXIgbW9kZWwgZXJyb3I7IGRlbGliZXJhdGVseSBkaWZmaWN1bHQsIG5vdCByZXByZXNlbnRhdGl2ZSBvZiB0eXBpY2FsIHVzYWdlLiBPcGVuQUkgcmVzZWFyY2ggZW52aXJvbm1lbnQgb3IgQVBJOyBzYW1lIG5hbWVkIG1ldHJpYyBhbmQgZXZhbHVhdGlvbiBjaGFydC4gTm8gZmFsbGJhY2sgaXMgcmVwb3J0ZWQgZm9yIHRoaXMgT3BlbkFJIGNvbmZpZ3VyYXRpb24u","id":"launch-openai-gpt61-sol-launch-2446","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"knowledge","comparabilityReason":"Private user-flagged prompt protocol needs separate review before capability scoring; retained as an exact lower-is-better sourced claim.","comparable":false,"configuration":"Reported reasoning effort xhigh. Share of answers with at least one factual error on de-identified conversations where users flagged an earlier model error; deliberately difficult, not representative of typical usage. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"Factual error rate on difficult prompts","locator":"Chart: Factual error rate on difficult prompts (lower is better) / GPT-6 Astra / xhigh","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $0.6032; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":3.99,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"factual-error-rate-on-difficult-prompts-sol-6-1-launch-user-flagged-conversations","evidenceKind":"lab_self_report","harnessId":"factual-error-rate-on-difficult-prompts-sol-6-1-launch-user-flagged-conversations:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIFNoYXJlIG9mIGFuc3dlcnMgd2l0aCBhdCBsZWFzdCBvbmUgZmFjdHVhbCBlcnJvciBvbiBkZS1pZGVudGlmaWVkIGNvbnZlcnNhdGlvbnMgd2hlcmUgdXNlcnMgZmxhZ2dlZCBhbiBlYXJsaWVyIG1vZGVsIGVycm9yOyBkZWxpYmVyYXRlbHkgZGlmZmljdWx0LCBub3QgcmVwcmVzZW50YXRpdmUgb2YgdHlwaWNhbCB1c2FnZS4gT3BlbkFJIHJlc2VhcmNoIGVudmlyb25tZW50IG9yIEFQSTsgc2FtZSBuYW1lZCBtZXRyaWMgYW5kIGV2YWx1YXRpb24gY2hhcnQuIE5vIGZhbGxiYWNrIGlzIHJlcG9ydGVkIGZvciB0aGlzIE9wZW5BSSBjb25maWd1cmF0aW9uLg","id":"launch-openai-gpt61-sol-launch-2447","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"knowledge","comparabilityReason":"Private user-flagged prompt protocol needs separate review before capability scoring; retained as an exact lower-is-better sourced claim.","comparable":false,"configuration":"Reported reasoning effort max. Share of answers with at least one factual error on de-identified conversations where users flagged an earlier model error; deliberately difficult, not representative of typical usage. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"Factual error rate on difficult prompts","locator":"Chart: Factual error rate on difficult prompts (lower is better) / GPT-6 Astra / max","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $0.7865; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":3.91,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"factual-error-rate-on-difficult-prompts-sol-6-1-launch-user-flagged-conversations","evidenceKind":"lab_self_report","harnessId":"factual-error-rate-on-difficult-prompts-sol-6-1-launch-user-flagged-conversations:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBsb3cuIFNoYXJlIG9mIGFuc3dlcnMgd2l0aCBhdCBsZWFzdCBvbmUgZmFjdHVhbCBlcnJvciBvbiBkZS1pZGVudGlmaWVkIGNvbnZlcnNhdGlvbnMgd2hlcmUgdXNlcnMgZmxhZ2dlZCBhbiBlYXJsaWVyIG1vZGVsIGVycm9yOyBkZWxpYmVyYXRlbHkgZGlmZmljdWx0LCBub3QgcmVwcmVzZW50YXRpdmUgb2YgdHlwaWNhbCB1c2FnZS4gT3BlbkFJIHJlc2VhcmNoIGVudmlyb25tZW50IG9yIEFQSTsgc2FtZSBuYW1lZCBtZXRyaWMgYW5kIGV2YWx1YXRpb24gY2hhcnQuIE5vIGZhbGxiYWNrIGlzIHJlcG9ydGVkIGZvciB0aGlzIE9wZW5BSSBjb25maWd1cmF0aW9uLg","id":"launch-openai-gpt61-sol-launch-2448","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6.1-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"knowledge","comparabilityReason":"Private user-flagged prompt protocol needs separate review before capability scoring; retained as an exact lower-is-better sourced claim.","comparable":false,"configuration":"Reported reasoning effort low. Share of answers with at least one factual error on de-identified conversations where users flagged an earlier model error; deliberately difficult, not representative of typical usage. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"Factual error rate on difficult prompts","locator":"Chart: Factual error rate on difficult prompts (lower is better) / GPT-6.1 Sol / low","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $0.0452; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":7.72,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"factual-error-rate-on-difficult-prompts-sol-6-1-launch-user-flagged-conversations","evidenceKind":"lab_self_report","harnessId":"factual-error-rate-on-difficult-prompts-sol-6-1-launch-user-flagged-conversations:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtZWRpdW0uIFNoYXJlIG9mIGFuc3dlcnMgd2l0aCBhdCBsZWFzdCBvbmUgZmFjdHVhbCBlcnJvciBvbiBkZS1pZGVudGlmaWVkIGNvbnZlcnNhdGlvbnMgd2hlcmUgdXNlcnMgZmxhZ2dlZCBhbiBlYXJsaWVyIG1vZGVsIGVycm9yOyBkZWxpYmVyYXRlbHkgZGlmZmljdWx0LCBub3QgcmVwcmVzZW50YXRpdmUgb2YgdHlwaWNhbCB1c2FnZS4gT3BlbkFJIHJlc2VhcmNoIGVudmlyb25tZW50IG9yIEFQSTsgc2FtZSBuYW1lZCBtZXRyaWMgYW5kIGV2YWx1YXRpb24gY2hhcnQuIE5vIGZhbGxiYWNrIGlzIHJlcG9ydGVkIGZvciB0aGlzIE9wZW5BSSBjb25maWd1cmF0aW9uLg","id":"launch-openai-gpt61-sol-launch-2449","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6.1-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"knowledge","comparabilityReason":"Private user-flagged prompt protocol needs separate review before capability scoring; retained as an exact lower-is-better sourced claim.","comparable":false,"configuration":"Reported reasoning effort medium. Share of answers with at least one factual error on de-identified conversations where users flagged an earlier model error; deliberately difficult, not representative of typical usage. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"Factual error rate on difficult prompts","locator":"Chart: Factual error rate on difficult prompts (lower is better) / GPT-6.1 Sol / medium","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $0.0558; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":6.29,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"factual-error-rate-on-difficult-prompts-sol-6-1-launch-user-flagged-conversations","evidenceKind":"lab_self_report","harnessId":"factual-error-rate-on-difficult-prompts-sol-6-1-launch-user-flagged-conversations:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBoaWdoLiBTaGFyZSBvZiBhbnN3ZXJzIHdpdGggYXQgbGVhc3Qgb25lIGZhY3R1YWwgZXJyb3Igb24gZGUtaWRlbnRpZmllZCBjb252ZXJzYXRpb25zIHdoZXJlIHVzZXJzIGZsYWdnZWQgYW4gZWFybGllciBtb2RlbCBlcnJvcjsgZGVsaWJlcmF0ZWx5IGRpZmZpY3VsdCwgbm90IHJlcHJlc2VudGF0aXZlIG9mIHR5cGljYWwgdXNhZ2UuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","id":"launch-openai-gpt61-sol-launch-2450","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6.1-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"knowledge","comparabilityReason":"Private user-flagged prompt protocol needs separate review before capability scoring; retained as an exact lower-is-better sourced claim.","comparable":false,"configuration":"Reported reasoning effort high. Share of answers with at least one factual error on de-identified conversations where users flagged an earlier model error; deliberately difficult, not representative of typical usage. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"Factual error rate on difficult prompts","locator":"Chart: Factual error rate on difficult prompts (lower is better) / GPT-6.1 Sol / high","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $0.0815; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":4.52,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"factual-error-rate-on-difficult-prompts-sol-6-1-launch-user-flagged-conversations","evidenceKind":"lab_self_report","harnessId":"factual-error-rate-on-difficult-prompts-sol-6-1-launch-user-flagged-conversations:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCB4aGlnaC4gU2hhcmUgb2YgYW5zd2VycyB3aXRoIGF0IGxlYXN0IG9uZSBmYWN0dWFsIGVycm9yIG9uIGRlLWlkZW50aWZpZWQgY29udmVyc2F0aW9ucyB3aGVyZSB1c2VycyBmbGFnZ2VkIGFuIGVhcmxpZXIgbW9kZWwgZXJyb3I7IGRlbGliZXJhdGVseSBkaWZmaWN1bHQsIG5vdCByZXByZXNlbnRhdGl2ZSBvZiB0eXBpY2FsIHVzYWdlLiBPcGVuQUkgcmVzZWFyY2ggZW52aXJvbm1lbnQgb3IgQVBJOyBzYW1lIG5hbWVkIG1ldHJpYyBhbmQgZXZhbHVhdGlvbiBjaGFydC4gTm8gZmFsbGJhY2sgaXMgcmVwb3J0ZWQgZm9yIHRoaXMgT3BlbkFJIGNvbmZpZ3VyYXRpb24u","id":"launch-openai-gpt61-sol-launch-2451","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6.1-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"knowledge","comparabilityReason":"Private user-flagged prompt protocol needs separate review before capability scoring; retained as an exact lower-is-better sourced claim.","comparable":false,"configuration":"Reported reasoning effort xhigh. Share of answers with at least one factual error on de-identified conversations where users flagged an earlier model error; deliberately difficult, not representative of typical usage. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"Factual error rate on difficult prompts","locator":"Chart: Factual error rate on difficult prompts (lower is better) / GPT-6.1 Sol / xhigh","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $0.0999; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":4.12,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"factual-error-rate-on-difficult-prompts-sol-6-1-launch-user-flagged-conversations","evidenceKind":"lab_self_report","harnessId":"factual-error-rate-on-difficult-prompts-sol-6-1-launch-user-flagged-conversations:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIFNoYXJlIG9mIGFuc3dlcnMgd2l0aCBhdCBsZWFzdCBvbmUgZmFjdHVhbCBlcnJvciBvbiBkZS1pZGVudGlmaWVkIGNvbnZlcnNhdGlvbnMgd2hlcmUgdXNlcnMgZmxhZ2dlZCBhbiBlYXJsaWVyIG1vZGVsIGVycm9yOyBkZWxpYmVyYXRlbHkgZGlmZmljdWx0LCBub3QgcmVwcmVzZW50YXRpdmUgb2YgdHlwaWNhbCB1c2FnZS4gT3BlbkFJIHJlc2VhcmNoIGVudmlyb25tZW50IG9yIEFQSTsgc2FtZSBuYW1lZCBtZXRyaWMgYW5kIGV2YWx1YXRpb24gY2hhcnQuIE5vIGZhbGxiYWNrIGlzIHJlcG9ydGVkIGZvciB0aGlzIE9wZW5BSSBjb25maWd1cmF0aW9uLg","id":"launch-openai-gpt61-sol-launch-2452","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6.1-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"knowledge","comparabilityReason":"Private user-flagged prompt protocol needs separate review before capability scoring; retained as an exact lower-is-better sourced claim.","comparable":false,"configuration":"Reported reasoning effort max. Share of answers with at least one factual error on de-identified conversations where users flagged an earlier model error; deliberately difficult, not representative of typical usage. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"Factual error rate on difficult prompts","locator":"Chart: Factual error rate on difficult prompts (lower is better) / GPT-6.1 Sol / max","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $0.1301; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":4.61,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"factual-error-rate-on-difficult-prompts-sol-6-1-launch-user-flagged-conversations","evidenceKind":"lab_self_report","harnessId":"factual-error-rate-on-difficult-prompts-sol-6-1-launch-user-flagged-conversations:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBsb3cuIFNoYXJlIG9mIGFuc3dlcnMgd2l0aCBhdCBsZWFzdCBvbmUgZmFjdHVhbCBlcnJvciBvbiBkZS1pZGVudGlmaWVkIGNvbnZlcnNhdGlvbnMgd2hlcmUgdXNlcnMgZmxhZ2dlZCBhbiBlYXJsaWVyIG1vZGVsIGVycm9yOyBkZWxpYmVyYXRlbHkgZGlmZmljdWx0LCBub3QgcmVwcmVzZW50YXRpdmUgb2YgdHlwaWNhbCB1c2FnZS4gT3BlbkFJIHJlc2VhcmNoIGVudmlyb25tZW50IG9yIEFQSTsgc2FtZSBuYW1lZCBtZXRyaWMgYW5kIGV2YWx1YXRpb24gY2hhcnQuIE5vIGZhbGxiYWNrIGlzIHJlcG9ydGVkIGZvciB0aGlzIE9wZW5BSSBjb25maWd1cmF0aW9uLg","id":"launch-openai-gpt61-sol-launch-2453","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"knowledge","comparabilityReason":"Private user-flagged prompt protocol needs separate review before capability scoring; retained as an exact lower-is-better sourced claim.","comparable":false,"configuration":"Reported reasoning effort low. Share of answers with at least one factual error on de-identified conversations where users flagged an earlier model error; deliberately difficult, not representative of typical usage. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"Factual error rate on difficult prompts","locator":"Chart: Factual error rate on difficult prompts (lower is better) / GPT-6 Sol / low","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $0.0495; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":11.42,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"factual-error-rate-on-difficult-prompts-sol-6-1-launch-user-flagged-conversations","evidenceKind":"lab_self_report","harnessId":"factual-error-rate-on-difficult-prompts-sol-6-1-launch-user-flagged-conversations:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtZWRpdW0uIFNoYXJlIG9mIGFuc3dlcnMgd2l0aCBhdCBsZWFzdCBvbmUgZmFjdHVhbCBlcnJvciBvbiBkZS1pZGVudGlmaWVkIGNvbnZlcnNhdGlvbnMgd2hlcmUgdXNlcnMgZmxhZ2dlZCBhbiBlYXJsaWVyIG1vZGVsIGVycm9yOyBkZWxpYmVyYXRlbHkgZGlmZmljdWx0LCBub3QgcmVwcmVzZW50YXRpdmUgb2YgdHlwaWNhbCB1c2FnZS4gT3BlbkFJIHJlc2VhcmNoIGVudmlyb25tZW50IG9yIEFQSTsgc2FtZSBuYW1lZCBtZXRyaWMgYW5kIGV2YWx1YXRpb24gY2hhcnQuIE5vIGZhbGxiYWNrIGlzIHJlcG9ydGVkIGZvciB0aGlzIE9wZW5BSSBjb25maWd1cmF0aW9uLg","id":"launch-openai-gpt61-sol-launch-2454","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"knowledge","comparabilityReason":"Private user-flagged prompt protocol needs separate review before capability scoring; retained as an exact lower-is-better sourced claim.","comparable":false,"configuration":"Reported reasoning effort medium. Share of answers with at least one factual error on de-identified conversations where users flagged an earlier model error; deliberately difficult, not representative of typical usage. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"Factual error rate on difficult prompts","locator":"Chart: Factual error rate on difficult prompts (lower is better) / GPT-6 Sol / medium","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $0.0694; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":6.86,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"factual-error-rate-on-difficult-prompts-sol-6-1-launch-user-flagged-conversations","evidenceKind":"lab_self_report","harnessId":"factual-error-rate-on-difficult-prompts-sol-6-1-launch-user-flagged-conversations:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBoaWdoLiBTaGFyZSBvZiBhbnN3ZXJzIHdpdGggYXQgbGVhc3Qgb25lIGZhY3R1YWwgZXJyb3Igb24gZGUtaWRlbnRpZmllZCBjb252ZXJzYXRpb25zIHdoZXJlIHVzZXJzIGZsYWdnZWQgYW4gZWFybGllciBtb2RlbCBlcnJvcjsgZGVsaWJlcmF0ZWx5IGRpZmZpY3VsdCwgbm90IHJlcHJlc2VudGF0aXZlIG9mIHR5cGljYWwgdXNhZ2UuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","id":"launch-openai-gpt61-sol-launch-2455","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"knowledge","comparabilityReason":"Private user-flagged prompt protocol needs separate review before capability scoring; retained as an exact lower-is-better sourced claim.","comparable":false,"configuration":"Reported reasoning effort high. Share of answers with at least one factual error on de-identified conversations where users flagged an earlier model error; deliberately difficult, not representative of typical usage. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"Factual error rate on difficult prompts","locator":"Chart: Factual error rate on difficult prompts (lower is better) / GPT-6 Sol / high","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $0.0994; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":5.14,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"factual-error-rate-on-difficult-prompts-sol-6-1-launch-user-flagged-conversations","evidenceKind":"lab_self_report","harnessId":"factual-error-rate-on-difficult-prompts-sol-6-1-launch-user-flagged-conversations:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCB4aGlnaC4gU2hhcmUgb2YgYW5zd2VycyB3aXRoIGF0IGxlYXN0IG9uZSBmYWN0dWFsIGVycm9yIG9uIGRlLWlkZW50aWZpZWQgY29udmVyc2F0aW9ucyB3aGVyZSB1c2VycyBmbGFnZ2VkIGFuIGVhcmxpZXIgbW9kZWwgZXJyb3I7IGRlbGliZXJhdGVseSBkaWZmaWN1bHQsIG5vdCByZXByZXNlbnRhdGl2ZSBvZiB0eXBpY2FsIHVzYWdlLiBPcGVuQUkgcmVzZWFyY2ggZW52aXJvbm1lbnQgb3IgQVBJOyBzYW1lIG5hbWVkIG1ldHJpYyBhbmQgZXZhbHVhdGlvbiBjaGFydC4gTm8gZmFsbGJhY2sgaXMgcmVwb3J0ZWQgZm9yIHRoaXMgT3BlbkFJIGNvbmZpZ3VyYXRpb24u","id":"launch-openai-gpt61-sol-launch-2456","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"knowledge","comparabilityReason":"Private user-flagged prompt protocol needs separate review before capability scoring; retained as an exact lower-is-better sourced claim.","comparable":false,"configuration":"Reported reasoning effort xhigh. Share of answers with at least one factual error on de-identified conversations where users flagged an earlier model error; deliberately difficult, not representative of typical usage. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"Factual error rate on difficult prompts","locator":"Chart: Factual error rate on difficult prompts (lower is better) / GPT-6 Sol / xhigh","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $0.1348; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":4.52,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"factual-error-rate-on-difficult-prompts-sol-6-1-launch-user-flagged-conversations","evidenceKind":"lab_self_report","harnessId":"factual-error-rate-on-difficult-prompts-sol-6-1-launch-user-flagged-conversations:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIFNoYXJlIG9mIGFuc3dlcnMgd2l0aCBhdCBsZWFzdCBvbmUgZmFjdHVhbCBlcnJvciBvbiBkZS1pZGVudGlmaWVkIGNvbnZlcnNhdGlvbnMgd2hlcmUgdXNlcnMgZmxhZ2dlZCBhbiBlYXJsaWVyIG1vZGVsIGVycm9yOyBkZWxpYmVyYXRlbHkgZGlmZmljdWx0LCBub3QgcmVwcmVzZW50YXRpdmUgb2YgdHlwaWNhbCB1c2FnZS4gT3BlbkFJIHJlc2VhcmNoIGVudmlyb25tZW50IG9yIEFQSTsgc2FtZSBuYW1lZCBtZXRyaWMgYW5kIGV2YWx1YXRpb24gY2hhcnQuIE5vIGZhbGxiYWNrIGlzIHJlcG9ydGVkIGZvciB0aGlzIE9wZW5BSSBjb25maWd1cmF0aW9uLg","id":"launch-openai-gpt61-sol-launch-2457","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"knowledge","comparabilityReason":"Private user-flagged prompt protocol needs separate review before capability scoring; retained as an exact lower-is-better sourced claim.","comparable":false,"configuration":"Reported reasoning effort max. Share of answers with at least one factual error on de-identified conversations where users flagged an earlier model error; deliberately difficult, not representative of typical usage. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"Factual error rate on difficult prompts","locator":"Chart: Factual error rate on difficult prompts (lower is better) / GPT-6 Sol / max","metric":"percent","notes":"Provider-published lab self-report. Published chart cost per task $0.179; source-specific cost does not establish consensus workload efficiency. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":4.57,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"failure-to-disclose-a-broken-search-tool-sol-6-1-launch-safety-stress-test","evidenceKind":"lab_self_report","harnessId":"failure-to-disclose-a-broken-search-tool-sol-6-1-launch-safety-stress-test:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIEFkdmVyc2FyaWFsIHRlc3Qgb2Ygd2hldGhlciBhZ2VudHMgZGlzY2xvc2UgYSBicm9rZW4gc2VhcmNoIHRvb2wuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","id":"launch-openai-gpt61-sol-launch-2458","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6.1-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Safety failure behavior is not a capability or intelligence score.","comparable":false,"configuration":"Reported reasoning effort max. Adversarial test of whether agents disclose a broken search tool. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"Failure to disclose a broken search tool","locator":"Chart: Failure to disclose a broken search tool (lower is better) / GPT-6.1 Sol / max","metric":"percent","notes":"Provider-published lab self-report. Safety stress outcomes are retained for inspection and excluded from capability scoring. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":2.1,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"failure-to-disclose-a-broken-search-tool-sol-6-1-launch-safety-stress-test","evidenceKind":"lab_self_report","harnessId":"failure-to-disclose-a-broken-search-tool-sol-6-1-launch-safety-stress-test:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIEFkdmVyc2FyaWFsIHRlc3Qgb2Ygd2hldGhlciBhZ2VudHMgZGlzY2xvc2UgYSBicm9rZW4gc2VhcmNoIHRvb2wuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","id":"launch-openai-gpt61-sol-launch-2459","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Safety failure behavior is not a capability or intelligence score.","comparable":false,"configuration":"Reported reasoning effort max. Adversarial test of whether agents disclose a broken search tool. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"Failure to disclose a broken search tool","locator":"Chart: Failure to disclose a broken search tool (lower is better) / GPT-6 Astra / max","metric":"percent","notes":"Provider-published lab self-report. Safety stress outcomes are retained for inspection and excluded from capability scoring. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":1.5,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"failure-to-disclose-a-broken-search-tool-sol-6-1-launch-safety-stress-test","evidenceKind":"lab_self_report","harnessId":"failure-to-disclose-a-broken-search-tool-sol-6-1-launch-safety-stress-test:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIEFkdmVyc2FyaWFsIHRlc3Qgb2Ygd2hldGhlciBhZ2VudHMgZGlzY2xvc2UgYSBicm9rZW4gc2VhcmNoIHRvb2wuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","id":"launch-openai-gpt61-sol-launch-2460","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Safety failure behavior is not a capability or intelligence score.","comparable":false,"configuration":"Reported reasoning effort max. Adversarial test of whether agents disclose a broken search tool. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"Failure to disclose a broken search tool","locator":"Chart: Failure to disclose a broken search tool (lower is better) / GPT-6 Sol / max","metric":"percent","notes":"Provider-published lab self-report. Safety stress outcomes are retained for inspection and excluded from capability scoring. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":4.92,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"failure-to-disclose-a-broken-search-tool-sol-6-1-launch-safety-stress-test","evidenceKind":"lab_self_report","harnessId":"failure-to-disclose-a-broken-search-tool-sol-6-1-launch-safety-stress-test:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIEFkdmVyc2FyaWFsIHRlc3Qgb2Ygd2hldGhlciBhZ2VudHMgZGlzY2xvc2UgYSBicm9rZW4gc2VhcmNoIHRvb2wuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","id":"launch-openai-gpt61-sol-launch-2461","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-luna","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Safety failure behavior is not a capability or intelligence score.","comparable":false,"configuration":"Reported reasoning effort max. Adversarial test of whether agents disclose a broken search tool. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"Failure to disclose a broken search tool","locator":"Chart: Failure to disclose a broken search tool (lower is better) / GPT-6 Luna / max","metric":"percent","notes":"Provider-published lab self-report. Safety stress outcomes are retained for inspection and excluded from capability scoring. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":28.67,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"reviewer-bypass-attempts-sol-6-1-launch-safety-stress-test","evidenceKind":"lab_self_report","harnessId":"reviewer-bypass-attempts-sol-6-1-launch-safety-stress-test:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIEF0dGVtcHRzIHRvIHdvcmsgYXJvdW5kIGFuIGF1dG9tYXRlZCBzYWZldHkgcmV2aWV3ZXIgYmxvY2tpbmcgYW4gYWN0aW9uLiBPcGVuQUkgcmVzZWFyY2ggZW52aXJvbm1lbnQgb3IgQVBJOyBzYW1lIG5hbWVkIG1ldHJpYyBhbmQgZXZhbHVhdGlvbiBjaGFydC4gTm8gZmFsbGJhY2sgaXMgcmVwb3J0ZWQgZm9yIHRoaXMgT3BlbkFJIGNvbmZpZ3VyYXRpb24u","id":"launch-openai-gpt61-sol-launch-2462","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6.1-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Safety failure behavior is not a capability or intelligence score.","comparable":false,"configuration":"Reported reasoning effort max. Attempts to work around an automated safety reviewer blocking an action. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"Reviewer bypass attempts","locator":"Chart: Reviewer bypass attempts (lower is better) / GPT-6.1 Sol / max","metric":"percent","notes":"Provider-published lab self-report. Safety stress outcomes are retained for inspection and excluded from capability scoring. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":0,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"reviewer-bypass-attempts-sol-6-1-launch-safety-stress-test","evidenceKind":"lab_self_report","harnessId":"reviewer-bypass-attempts-sol-6-1-launch-safety-stress-test:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIEF0dGVtcHRzIHRvIHdvcmsgYXJvdW5kIGFuIGF1dG9tYXRlZCBzYWZldHkgcmV2aWV3ZXIgYmxvY2tpbmcgYW4gYWN0aW9uLiBPcGVuQUkgcmVzZWFyY2ggZW52aXJvbm1lbnQgb3IgQVBJOyBzYW1lIG5hbWVkIG1ldHJpYyBhbmQgZXZhbHVhdGlvbiBjaGFydC4gTm8gZmFsbGJhY2sgaXMgcmVwb3J0ZWQgZm9yIHRoaXMgT3BlbkFJIGNvbmZpZ3VyYXRpb24u","id":"launch-openai-gpt61-sol-launch-2463","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Safety failure behavior is not a capability or intelligence score.","comparable":false,"configuration":"Reported reasoning effort max. Attempts to work around an automated safety reviewer blocking an action. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"Reviewer bypass attempts","locator":"Chart: Reviewer bypass attempts (lower is better) / GPT-6 Sol / max","metric":"percent","notes":"Provider-published lab self-report. Safety stress outcomes are retained for inspection and excluded from capability scoring. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":0,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"reviewer-bypass-attempts-sol-6-1-launch-safety-stress-test","evidenceKind":"lab_self_report","harnessId":"reviewer-bypass-attempts-sol-6-1-launch-safety-stress-test:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIEF0dGVtcHRzIHRvIHdvcmsgYXJvdW5kIGFuIGF1dG9tYXRlZCBzYWZldHkgcmV2aWV3ZXIgYmxvY2tpbmcgYW4gYWN0aW9uLiBPcGVuQUkgcmVzZWFyY2ggZW52aXJvbm1lbnQgb3IgQVBJOyBzYW1lIG5hbWVkIG1ldHJpYyBhbmQgZXZhbHVhdGlvbiBjaGFydC4gTm8gZmFsbGJhY2sgaXMgcmVwb3J0ZWQgZm9yIHRoaXMgT3BlbkFJIGNvbmZpZ3VyYXRpb24u","id":"launch-openai-gpt61-sol-launch-2464","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-luna","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Safety failure behavior is not a capability or intelligence score.","comparable":false,"configuration":"Reported reasoning effort max. Attempts to work around an automated safety reviewer blocking an action. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"Reviewer bypass attempts","locator":"Chart: Reviewer bypass attempts (lower is better) / GPT-6 Luna / max","metric":"percent","notes":"Provider-published lab self-report. Safety stress outcomes are retained for inspection and excluded from capability scoring. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":0.26,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"reviewer-bypass-attempts-sol-6-1-launch-safety-stress-test","evidenceKind":"lab_self_report","harnessId":"reviewer-bypass-attempts-sol-6-1-launch-safety-stress-test:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIEF0dGVtcHRzIHRvIHdvcmsgYXJvdW5kIGFuIGF1dG9tYXRlZCBzYWZldHkgcmV2aWV3ZXIgYmxvY2tpbmcgYW4gYWN0aW9uLiBPcGVuQUkgcmVzZWFyY2ggZW52aXJvbm1lbnQgb3IgQVBJOyBzYW1lIG5hbWVkIG1ldHJpYyBhbmQgZXZhbHVhdGlvbiBjaGFydC4gTm8gZmFsbGJhY2sgaXMgcmVwb3J0ZWQgZm9yIHRoaXMgT3BlbkFJIGNvbmZpZ3VyYXRpb24u","id":"launch-openai-gpt61-sol-launch-2465","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Safety failure behavior is not a capability or intelligence score.","comparable":false,"configuration":"Reported reasoning effort max. Attempts to work around an automated safety reviewer blocking an action. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"Reviewer bypass attempts","locator":"Chart: Reviewer bypass attempts (lower is better) / GPT-6 Astra / max","metric":"percent","notes":"Provider-published lab self-report. Safety stress outcomes are retained for inspection and excluded from capability scoring. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":0,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"warning-circumvention-sol-6-1-launch-safety-stress-test","evidenceKind":"lab_self_report","harnessId":"warning-circumvention-sol-6-1-launch-safety-stress-test:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIFJlc3RyaWN0aW9uLWNpcmN1bXZlbnRpb24gYXR0ZW1wdHMgaW4gZGVsaWJlcmF0ZWx5IGRpZmZpY3VsdCBsb3ctc3Rha2VzIGNhc2VzIHdpdGhvdXQgZnVsbCBwcm9kdWN0IHNhZmVndWFyZHMuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","id":"launch-openai-gpt61-sol-launch-2466","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6.1-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Safety failure behavior is not a capability or intelligence score.","comparable":false,"configuration":"Reported reasoning effort max. Restriction-circumvention attempts in deliberately difficult low-stakes cases without full product safeguards. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"Warning circumvention","locator":"Chart: Warning circumvention (lower is better) / GPT-6.1 Sol / max","metric":"percent","notes":"Provider-published lab self-report. Safety stress outcomes are retained for inspection and excluded from capability scoring. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":23.48,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"warning-circumvention-sol-6-1-launch-safety-stress-test","evidenceKind":"lab_self_report","harnessId":"warning-circumvention-sol-6-1-launch-safety-stress-test:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIFJlc3RyaWN0aW9uLWNpcmN1bXZlbnRpb24gYXR0ZW1wdHMgaW4gZGVsaWJlcmF0ZWx5IGRpZmZpY3VsdCBsb3ctc3Rha2VzIGNhc2VzIHdpdGhvdXQgZnVsbCBwcm9kdWN0IHNhZmVndWFyZHMuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","id":"launch-openai-gpt61-sol-launch-2467","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Safety failure behavior is not a capability or intelligence score.","comparable":false,"configuration":"Reported reasoning effort max. Restriction-circumvention attempts in deliberately difficult low-stakes cases without full product safeguards. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"Warning circumvention","locator":"Chart: Warning circumvention (lower is better) / GPT-6 Sol / max","metric":"percent","notes":"Provider-published lab self-report. Safety stress outcomes are retained for inspection and excluded from capability scoring. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":64.39,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"warning-circumvention-sol-6-1-launch-safety-stress-test","evidenceKind":"lab_self_report","harnessId":"warning-circumvention-sol-6-1-launch-safety-stress-test:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIFJlc3RyaWN0aW9uLWNpcmN1bXZlbnRpb24gYXR0ZW1wdHMgaW4gZGVsaWJlcmF0ZWx5IGRpZmZpY3VsdCBsb3ctc3Rha2VzIGNhc2VzIHdpdGhvdXQgZnVsbCBwcm9kdWN0IHNhZmVndWFyZHMuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","id":"launch-openai-gpt61-sol-launch-2468","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-luna","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Safety failure behavior is not a capability or intelligence score.","comparable":false,"configuration":"Reported reasoning effort max. Restriction-circumvention attempts in deliberately difficult low-stakes cases without full product safeguards. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"Warning circumvention","locator":"Chart: Warning circumvention (lower is better) / GPT-6 Luna / max","metric":"percent","notes":"Provider-published lab self-report. Safety stress outcomes are retained for inspection and excluded from capability scoring. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":42.37,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"warning-circumvention-sol-6-1-launch-safety-stress-test","evidenceKind":"lab_self_report","harnessId":"warning-circumvention-sol-6-1-launch-safety-stress-test:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCBtYXguIFJlc3RyaWN0aW9uLWNpcmN1bXZlbnRpb24gYXR0ZW1wdHMgaW4gZGVsaWJlcmF0ZWx5IGRpZmZpY3VsdCBsb3ctc3Rha2VzIGNhc2VzIHdpdGhvdXQgZnVsbCBwcm9kdWN0IHNhZmVndWFyZHMuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","id":"launch-openai-gpt61-sol-launch-2469","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Safety failure behavior is not a capability or intelligence score.","comparable":false,"configuration":"Reported reasoning effort max. Restriction-circumvention attempts in deliberately difficult low-stakes cases without full product safeguards. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"Warning circumvention","locator":"Chart: Warning circumvention (lower is better) / GPT-6 Astra / max","metric":"percent","notes":"Provider-published lab self-report. Safety stress outcomes are retained for inspection and excluded from capability scoring. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":17.42,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"computer-use-safety-stress-test-sol-6-1-launch-safety-stress-test","evidenceKind":"lab_self_report","harnessId":"computer-use-safety-stress-test-sol-6-1-launch-safety-stress-test:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCB4aGlnaC4gVW5pbnRlbmRlZCBvdXRjb21lcyBpbiBkZWxpYmVyYXRlbHkgYWR2ZXJzYXJpYWwgY29tcHV0ZXItIGFuZCBicm93c2VyLXVzZSB3b3JrcGxhY2UgdGFza3M7IHRoZSB1cGRhdGVkIGhhcmRlciBzYWZldHkgc3Vic2V0LCBub3QgT1NXb3JsZCB0YXNrIHN1Y2Nlc3MuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","id":"launch-openai-gpt61-sol-launch-2470","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6.1-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Safety failure behavior is not a capability or intelligence score.","comparable":false,"configuration":"Reported reasoning effort xhigh. Unintended outcomes in deliberately adversarial computer- and browser-use workplace tasks; the updated harder safety subset, not OSWorld task success. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"Computer-use safety stress test","locator":"Chart: Computer-use safety stress test (lower is better) / GPT-6.1 Sol / xhigh","metric":"percent","notes":"Provider-published lab self-report. Safety stress outcomes are retained for inspection and excluded from capability scoring. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":4.32,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"computer-use-safety-stress-test-sol-6-1-launch-safety-stress-test","evidenceKind":"lab_self_report","harnessId":"computer-use-safety-stress-test-sol-6-1-launch-safety-stress-test:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCB4aGlnaC4gVW5pbnRlbmRlZCBvdXRjb21lcyBpbiBkZWxpYmVyYXRlbHkgYWR2ZXJzYXJpYWwgY29tcHV0ZXItIGFuZCBicm93c2VyLXVzZSB3b3JrcGxhY2UgdGFza3M7IHRoZSB1cGRhdGVkIGhhcmRlciBzYWZldHkgc3Vic2V0LCBub3QgT1NXb3JsZCB0YXNrIHN1Y2Nlc3MuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","id":"launch-openai-gpt61-sol-launch-2471","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-sol","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Safety failure behavior is not a capability or intelligence score.","comparable":false,"configuration":"Reported reasoning effort xhigh. Unintended outcomes in deliberately adversarial computer- and browser-use workplace tasks; the updated harder safety subset, not OSWorld task success. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"Computer-use safety stress test","locator":"Chart: Computer-use safety stress test (lower is better) / GPT-6 Sol / xhigh","metric":"percent","notes":"Provider-published lab self-report. Safety stress outcomes are retained for inspection and excluded from capability scoring. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":17.39,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"computer-use-safety-stress-test-sol-6-1-launch-safety-stress-test","evidenceKind":"lab_self_report","harnessId":"computer-use-safety-stress-test-sol-6-1-launch-safety-stress-test:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCB4aGlnaC4gVW5pbnRlbmRlZCBvdXRjb21lcyBpbiBkZWxpYmVyYXRlbHkgYWR2ZXJzYXJpYWwgY29tcHV0ZXItIGFuZCBicm93c2VyLXVzZSB3b3JrcGxhY2UgdGFza3M7IHRoZSB1cGRhdGVkIGhhcmRlciBzYWZldHkgc3Vic2V0LCBub3QgT1NXb3JsZCB0YXNrIHN1Y2Nlc3MuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","id":"launch-openai-gpt61-sol-launch-2472","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-luna","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Safety failure behavior is not a capability or intelligence score.","comparable":false,"configuration":"Reported reasoning effort xhigh. Unintended outcomes in deliberately adversarial computer- and browser-use workplace tasks; the updated harder safety subset, not OSWorld task success. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"Computer-use safety stress test","locator":"Chart: Computer-use safety stress test (lower is better) / GPT-6 Luna / xhigh","metric":"percent","notes":"Provider-published lab self-report. Safety stress outcomes are retained for inspection and excluded from capability scoring. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":13.7,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"computer-use-safety-stress-test-sol-6-1-launch-safety-stress-test","evidenceKind":"lab_self_report","harnessId":"computer-use-safety-stress-test-sol-6-1-launch-safety-stress-test:openai-gpt61-sol-launch:UmVwb3J0ZWQgcmVhc29uaW5nIGVmZm9ydCB4aGlnaC4gVW5pbnRlbmRlZCBvdXRjb21lcyBpbiBkZWxpYmVyYXRlbHkgYWR2ZXJzYXJpYWwgY29tcHV0ZXItIGFuZCBicm93c2VyLXVzZSB3b3JrcGxhY2UgdGFza3M7IHRoZSB1cGRhdGVkIGhhcmRlciBzYWZldHkgc3Vic2V0LCBub3QgT1NXb3JsZCB0YXNrIHN1Y2Nlc3MuIE9wZW5BSSByZXNlYXJjaCBlbnZpcm9ubWVudCBvciBBUEk7IHNhbWUgbmFtZWQgbWV0cmljIGFuZCBldmFsdWF0aW9uIGNoYXJ0LiBObyBmYWxsYmFjayBpcyByZXBvcnRlZCBmb3IgdGhpcyBPcGVuQUkgY29uZmlndXJhdGlvbi4","id":"launch-openai-gpt61-sol-launch-2473","ingestRunId":"launch-openai-gpt61-sol-launch","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-29","rawPayloadHash":"0786cd9dcca56da778589fa71567360532763dbfa10f46f128e421a21a6e791e","report":{"category":"agentic","comparabilityReason":"Safety failure behavior is not a capability or intelligence score.","comparable":false,"configuration":"Reported reasoning effort xhigh. Unintended outcomes in deliberately adversarial computer- and browser-use workplace tasks; the updated harder safety subset, not OSWorld task success. OpenAI research environment or API; same named metric and evaluation chart. No fallback is reported for this OpenAI configuration.","family":"Computer-use safety stress test","locator":"Chart: Computer-use safety stress test (lower is better) / GPT-6 Astra / xhigh","metric":"percent","notes":"Provider-published lab self-report. Safety stress outcomes are retained for inspection and excluded from capability scoring. Competitor results are republished from public reports; the page does not establish independent reproduction by OpenAI.","sourceId":"openai-gpt61-sol-launch","sourceTitle":"Introducing GPT-6.1 Sol"},"score":2.4,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/introducing-gpt-6-1-sol/"},{"benchmarkId":"agents-last-exam-not-specified","evidenceKind":"lab_self_report","harnessId":"agents-last-exam-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-269","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"Agents' Last Exam","locator":"Professional table / Agents' Last Exam / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":52.7,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"agents-last-exam-not-specified","evidenceKind":"lab_self_report","harnessId":"agents-last-exam-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-270","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"Agents' Last Exam","locator":"Professional table / Agents' Last Exam / GPT‑5.6 Terra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":50.4,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"agents-last-exam-not-specified","evidenceKind":"lab_self_report","harnessId":"agents-last-exam-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-271","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-luna","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"Agents' Last Exam","locator":"Professional table / Agents' Last Exam / GPT‑5.6 Luna","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":50.3,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"agents-last-exam-not-specified","evidenceKind":"lab_self_report","harnessId":"agents-last-exam-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-272","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"Agents' Last Exam","locator":"Professional table / Agents' Last Exam / GPT‑5.5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":46.9,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"agents-last-exam-not-specified","evidenceKind":"lab_self_report","harnessId":"agents-last-exam-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-273","ingestRunId":"launch-openai-sol-launch","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"Agents' Last Exam","locator":"Professional table / Agents' Last Exam / Claude Fable 5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":40.5,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"agents-last-exam-not-specified","evidenceKind":"lab_self_report","harnessId":"agents-last-exam-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-274","ingestRunId":"launch-openai-sol-launch","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"Agents' Last Exam","locator":"Professional table / Agents' Last Exam / Claude Opus 4.8","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":45.2,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"agents-last-exam-not-specified","evidenceKind":"lab_self_report","harnessId":"agents-last-exam-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-275","ingestRunId":"launch-openai-sol-launch","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"Agents' Last Exam","locator":"Professional table / Agents' Last Exam / Gemini 3.1 Pro Preview","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":32.1,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"gdpval-aa-v2-2","evidenceKind":"lab_self_report","harnessId":"gdpval-aa-v2-2:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-276","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"GDPval-AA v2","locator":"Professional table / GDPval-AA v2 / GPT‑5.6 Sol","metric":"Elo","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":1747.8,"scoreUnit":"elo","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"gdpval-aa-v2-2","evidenceKind":"lab_self_report","harnessId":"gdpval-aa-v2-2:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-277","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"GDPval-AA v2","locator":"Professional table / GDPval-AA v2 / GPT‑5.6 Terra","metric":"Elo","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":1593,"scoreUnit":"elo","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"gdpval-aa-v2-2","evidenceKind":"lab_self_report","harnessId":"gdpval-aa-v2-2:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-278","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-luna","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"GDPval-AA v2","locator":"Professional table / GDPval-AA v2 / GPT‑5.6 Luna","metric":"Elo","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":1591.8,"scoreUnit":"elo","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"gdpval-aa-v2-2","evidenceKind":"lab_self_report","harnessId":"gdpval-aa-v2-2:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-279","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"GDPval-AA v2","locator":"Professional table / GDPval-AA v2 / GPT‑5.5","metric":"Elo","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":1493.7,"scoreUnit":"elo","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"gdpval-aa-v2-2","evidenceKind":"lab_self_report","harnessId":"gdpval-aa-v2-2:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-280","ingestRunId":"launch-openai-sol-launch","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"GDPval-AA v2","locator":"Professional table / GDPval-AA v2 / Claude Fable 5","metric":"Elo","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":1759.6,"scoreUnit":"elo","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"gdpval-aa-v2-2","evidenceKind":"lab_self_report","harnessId":"gdpval-aa-v2-2:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-281","ingestRunId":"launch-openai-sol-launch","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"GDPval-AA v2","locator":"Professional table / GDPval-AA v2 / Claude Opus 4.8","metric":"Elo","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":1600.1,"scoreUnit":"elo","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"gdpval-aa-v2-2","evidenceKind":"lab_self_report","harnessId":"gdpval-aa-v2-2:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-282","ingestRunId":"launch-openai-sol-launch","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"GDPval-AA v2","locator":"Professional table / GDPval-AA v2 / Gemini 3.1 Pro Preview","metric":"Elo","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":962.3,"scoreUnit":"elo","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"gdpval-aa-v2-2","evidenceKind":"lab_self_report","harnessId":"gdpval-aa-v2-2:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-283","ingestRunId":"launch-openai-sol-launch","modelId":"gemini-3.5-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"GDPval-AA v2","locator":"Professional table / GDPval-AA v2 / Gemini 3.5 Flash","metric":"Elo","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":1348.8,"scoreUnit":"elo","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"management-consulting-tasks-internal-not-specified","evidenceKind":"lab_self_report","harnessId":"management-consulting-tasks-internal-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-284","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"Management Consulting Tasks (Internal)","locator":"Professional table / Management Consulting Tasks (Internal) / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":43.2,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"management-consulting-tasks-internal-not-specified","evidenceKind":"lab_self_report","harnessId":"management-consulting-tasks-internal-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-285","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"Management Consulting Tasks (Internal)","locator":"Professional table / Management Consulting Tasks (Internal) / GPT‑5.6 Terra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":37.2,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"management-consulting-tasks-internal-not-specified","evidenceKind":"lab_self_report","harnessId":"management-consulting-tasks-internal-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-286","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-luna","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"Management Consulting Tasks (Internal)","locator":"Professional table / Management Consulting Tasks (Internal) / GPT‑5.6 Luna","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":35.4,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"management-consulting-tasks-internal-not-specified","evidenceKind":"lab_self_report","harnessId":"management-consulting-tasks-internal-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-287","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"Management Consulting Tasks (Internal)","locator":"Professional table / Management Consulting Tasks (Internal) / GPT‑5.5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":31.3,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"management-consulting-tasks-internal-not-specified","evidenceKind":"lab_self_report","harnessId":"management-consulting-tasks-internal-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-288","ingestRunId":"launch-openai-sol-launch","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"Management Consulting Tasks (Internal)","locator":"Professional table / Management Consulting Tasks (Internal) / Claude Fable 5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":35.5,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"management-consulting-tasks-internal-not-specified","evidenceKind":"lab_self_report","harnessId":"management-consulting-tasks-internal-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-289","ingestRunId":"launch-openai-sol-launch","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"Management Consulting Tasks (Internal)","locator":"Professional table / Management Consulting Tasks (Internal) / Claude Opus 4.8","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":31.6,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"management-consulting-tasks-internal-not-specified","evidenceKind":"lab_self_report","harnessId":"management-consulting-tasks-internal-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-290","ingestRunId":"launch-openai-sol-launch","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"Management Consulting Tasks (Internal)","locator":"Professional table / Management Consulting Tasks (Internal) / Gemini 3.1 Pro Preview","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":13.2,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"big-finance-bench-not-specified","evidenceKind":"lab_self_report","harnessId":"big-finance-bench-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-291","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"Big Finance Bench","locator":"Professional table / Big Finance Bench / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":53,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"big-finance-bench-not-specified","evidenceKind":"lab_self_report","harnessId":"big-finance-bench-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-292","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"Big Finance Bench","locator":"Professional table / Big Finance Bench / GPT‑5.6 Terra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":51,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"big-finance-bench-not-specified","evidenceKind":"lab_self_report","harnessId":"big-finance-bench-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-293","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-luna","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"Big Finance Bench","locator":"Professional table / Big Finance Bench / GPT‑5.6 Luna","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":36,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"big-finance-bench-not-specified","evidenceKind":"lab_self_report","harnessId":"big-finance-bench-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-294","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"Big Finance Bench","locator":"Professional table / Big Finance Bench / GPT‑5.5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":49,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"big-finance-bench-not-specified","evidenceKind":"lab_self_report","harnessId":"big-finance-bench-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-295","ingestRunId":"launch-openai-sol-launch","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"Big Finance Bench","locator":"Professional table / Big Finance Bench / Claude Opus 4.8","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":44,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"artificial-analysis-intelligence-index-v4-1-4-1","evidenceKind":"lab_self_report","harnessId":"artificial-analysis-intelligence-index-v4-1-4-1:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-296","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"Artificial Analysis Intelligence Index v4.1","locator":"Professional table / Artificial Analysis Intelligence Index v4.1 / GPT‑5.6 Sol","metric":"index score","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":58.9,"scoreUnit":"index","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"artificial-analysis-intelligence-index-v4-1-4-1","evidenceKind":"lab_self_report","harnessId":"artificial-analysis-intelligence-index-v4-1-4-1:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-297","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"Artificial Analysis Intelligence Index v4.1","locator":"Professional table / Artificial Analysis Intelligence Index v4.1 / GPT‑5.6 Terra","metric":"index score","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":55,"scoreUnit":"index","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"artificial-analysis-intelligence-index-v4-1-4-1","evidenceKind":"lab_self_report","harnessId":"artificial-analysis-intelligence-index-v4-1-4-1:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-298","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-luna","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"Artificial Analysis Intelligence Index v4.1","locator":"Professional table / Artificial Analysis Intelligence Index v4.1 / GPT‑5.6 Luna","metric":"index score","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":51.2,"scoreUnit":"index","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"artificial-analysis-intelligence-index-v4-1-4-1","evidenceKind":"lab_self_report","harnessId":"artificial-analysis-intelligence-index-v4-1-4-1:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-299","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"Artificial Analysis Intelligence Index v4.1","locator":"Professional table / Artificial Analysis Intelligence Index v4.1 / GPT‑5.5","metric":"index score","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":54.8,"scoreUnit":"index","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"artificial-analysis-intelligence-index-v4-1-4-1","evidenceKind":"lab_self_report","harnessId":"artificial-analysis-intelligence-index-v4-1-4-1:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-300","ingestRunId":"launch-openai-sol-launch","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"Artificial Analysis Intelligence Index v4.1","locator":"Professional table / Artificial Analysis Intelligence Index v4.1 / Claude Fable 5","metric":"index score","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":59.9,"scoreUnit":"index","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"artificial-analysis-intelligence-index-v4-1-4-1","evidenceKind":"lab_self_report","harnessId":"artificial-analysis-intelligence-index-v4-1-4-1:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-301","ingestRunId":"launch-openai-sol-launch","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"Artificial Analysis Intelligence Index v4.1","locator":"Professional table / Artificial Analysis Intelligence Index v4.1 / Claude Opus 4.8","metric":"index score","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":55.7,"scoreUnit":"index","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"artificial-analysis-intelligence-index-v4-1-4-1","evidenceKind":"lab_self_report","harnessId":"artificial-analysis-intelligence-index-v4-1-4-1:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-302","ingestRunId":"launch-openai-sol-launch","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"Artificial Analysis Intelligence Index v4.1","locator":"Professional table / Artificial Analysis Intelligence Index v4.1 / Gemini 3.1 Pro Preview","metric":"index score","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":46.5,"scoreUnit":"index","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"artificial-analysis-intelligence-index-v4-1-4-1","evidenceKind":"lab_self_report","harnessId":"artificial-analysis-intelligence-index-v4-1-4-1:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-303","ingestRunId":"launch-openai-sol-launch","modelId":"gemini-3.5-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"Artificial Analysis Intelligence Index v4.1","locator":"Professional table / Artificial Analysis Intelligence Index v4.1 / Gemini 3.5 Flash","metric":"index score","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":50.2,"scoreUnit":"index","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"artificial-analysis-coding-agent-index-v1-1-1-1","evidenceKind":"lab_self_report","harnessId":"artificial-analysis-coding-agent-index-v1-1-1-1:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-304","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"Artificial Analysis Coding Agent Index v1.1","locator":"Coding table / Artificial Analysis Coding Agent Index v1.1 / GPT‑5.6 Sol","metric":"index score","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":80,"scoreUnit":"index","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"artificial-analysis-coding-agent-index-v1-1-1-1","evidenceKind":"lab_self_report","harnessId":"artificial-analysis-coding-agent-index-v1-1-1-1:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-305","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"Artificial Analysis Coding Agent Index v1.1","locator":"Coding table / Artificial Analysis Coding Agent Index v1.1 / GPT‑5.6 Terra","metric":"index score","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":77.4,"scoreUnit":"index","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"artificial-analysis-coding-agent-index-v1-1-1-1","evidenceKind":"lab_self_report","harnessId":"artificial-analysis-coding-agent-index-v1-1-1-1:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-306","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-luna","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"Artificial Analysis Coding Agent Index v1.1","locator":"Coding table / Artificial Analysis Coding Agent Index v1.1 / GPT‑5.6 Luna","metric":"index score","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":74.6,"scoreUnit":"index","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"artificial-analysis-coding-agent-index-v1-1-1-1","evidenceKind":"lab_self_report","harnessId":"artificial-analysis-coding-agent-index-v1-1-1-1:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-307","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"Artificial Analysis Coding Agent Index v1.1","locator":"Coding table / Artificial Analysis Coding Agent Index v1.1 / GPT‑5.5","metric":"index score","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":76.4,"scoreUnit":"index","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"artificial-analysis-coding-agent-index-v1-1-1-1","evidenceKind":"lab_self_report","harnessId":"artificial-analysis-coding-agent-index-v1-1-1-1:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-308","ingestRunId":"launch-openai-sol-launch","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"Artificial Analysis Coding Agent Index v1.1","locator":"Coding table / Artificial Analysis Coding Agent Index v1.1 / Claude Fable 5","metric":"index score","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":77.2,"scoreUnit":"index","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"artificial-analysis-coding-agent-index-v1-1-1-1","evidenceKind":"lab_self_report","harnessId":"artificial-analysis-coding-agent-index-v1-1-1-1:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-309","ingestRunId":"launch-openai-sol-launch","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"Artificial Analysis Coding Agent Index v1.1","locator":"Coding table / Artificial Analysis Coding Agent Index v1.1 / Claude Opus 4.8","metric":"index score","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":72.5,"scoreUnit":"index","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"artificial-analysis-coding-agent-index-v1-1-1-1","evidenceKind":"lab_self_report","harnessId":"artificial-analysis-coding-agent-index-v1-1-1-1:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-310","ingestRunId":"launch-openai-sol-launch","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"Artificial Analysis Coding Agent Index v1.1","locator":"Coding table / Artificial Analysis Coding Agent Index v1.1 / Gemini 3.1 Pro Preview","metric":"index score","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":42.7,"scoreUnit":"index","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"swe-bench-pro-not-specified","evidenceKind":"lab_self_report","harnessId":"swe-bench-pro-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-311","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"SWE-bench Pro","locator":"Coding table / SWE-Bench Pro / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":64.6,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"swe-bench-pro-not-specified","evidenceKind":"lab_self_report","harnessId":"swe-bench-pro-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-312","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"SWE-bench Pro","locator":"Coding table / SWE-Bench Pro / GPT‑5.6 Terra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":63.4,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"swe-bench-pro-not-specified","evidenceKind":"lab_self_report","harnessId":"swe-bench-pro-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-313","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-luna","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"SWE-bench Pro","locator":"Coding table / SWE-Bench Pro / GPT‑5.6 Luna","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":62.7,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"swe-bench-pro-not-specified","evidenceKind":"lab_self_report","harnessId":"swe-bench-pro-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-314","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"SWE-bench Pro","locator":"Coding table / SWE-Bench Pro / GPT‑5.5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":59.4,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"swe-bench-pro-not-specified","evidenceKind":"lab_self_report","harnessId":"swe-bench-pro-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-315","ingestRunId":"launch-openai-sol-launch","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"SWE-bench Pro","locator":"Coding table / SWE-Bench Pro / Claude Fable 5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":80,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"swe-bench-pro-not-specified","evidenceKind":"lab_self_report","harnessId":"swe-bench-pro-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-316","ingestRunId":"launch-openai-sol-launch","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"SWE-bench Pro","locator":"Coding table / SWE-Bench Pro / Claude Opus 4.8","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":69.2,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"swe-bench-pro-not-specified","evidenceKind":"lab_self_report","harnessId":"swe-bench-pro-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-317","ingestRunId":"launch-openai-sol-launch","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"SWE-bench Pro","locator":"Coding table / SWE-Bench Pro / Gemini 3.1 Pro Preview","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":54.2,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"deepswe-v1-1-1-1-percent","evidenceKind":"lab_self_report","harnessId":"deepswe-v1-1-1-1-percent:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-318","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"DeepSWE v1.1","locator":"Coding table / DeepSWE v1.1 / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":72.7,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"deepswe-v1-1-1-1-percent","evidenceKind":"lab_self_report","harnessId":"deepswe-v1-1-1-1-percent:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-319","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"DeepSWE v1.1","locator":"Coding table / DeepSWE v1.1 / GPT‑5.6 Terra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":69.6,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"deepswe-v1-1-1-1-percent","evidenceKind":"lab_self_report","harnessId":"deepswe-v1-1-1-1-percent:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-320","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-luna","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"DeepSWE v1.1","locator":"Coding table / DeepSWE v1.1 / GPT‑5.6 Luna","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":67.2,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"deepswe-v1-1-1-1-percent","evidenceKind":"lab_self_report","harnessId":"deepswe-v1-1-1-1-percent:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-321","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"DeepSWE v1.1","locator":"Coding table / DeepSWE v1.1 / GPT‑5.5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":67,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"deepswe-v1-1-1-1-percent","evidenceKind":"lab_self_report","harnessId":"deepswe-v1-1-1-1-percent:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-322","ingestRunId":"launch-openai-sol-launch","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"DeepSWE v1.1","locator":"Coding table / DeepSWE v1.1 / Claude Fable 5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":69.7,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"deepswe-v1-1-1-1-percent","evidenceKind":"lab_self_report","harnessId":"deepswe-v1-1-1-1-percent:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-323","ingestRunId":"launch-openai-sol-launch","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"DeepSWE v1.1","locator":"Coding table / DeepSWE v1.1 / Claude Opus 4.8","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":59,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"deepswe-v1-1-1-1-percent","evidenceKind":"lab_self_report","harnessId":"deepswe-v1-1-1-1-percent:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-324","ingestRunId":"launch-openai-sol-launch","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"DeepSWE v1.1","locator":"Coding table / DeepSWE v1.1 / Gemini 3.1 Pro Preview","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":11.8,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"terminal-bench-2-1-percent","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1-percent:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-325","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"Terminal-Bench","locator":"Coding table / Terminal-Bench 2.1 / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":88.8,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"terminal-bench-2-1-percent","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1-percent:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IFVsdHJhLCBmb3VyLWFnZW50IG9yY2hlc3RyYXRpb24","id":"launch-openai-sol-launch-326","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; Ultra, four-agent orchestration","family":"Terminal-Bench","locator":"Coding table / Terminal-Bench 2.1 / GPT‑5.6 Sol Ultra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":91.9,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"terminal-bench-2-1-percent","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1-percent:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-327","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"Terminal-Bench","locator":"Coding table / Terminal-Bench 2.1 / GPT‑5.6 Terra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":87.4,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"terminal-bench-2-1-percent","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1-percent:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-328","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-luna","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"Terminal-Bench","locator":"Coding table / Terminal-Bench 2.1 / GPT‑5.6 Luna","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":84.7,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"terminal-bench-2-1-percent","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1-percent:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-329","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"Terminal-Bench","locator":"Coding table / Terminal-Bench 2.1 / GPT‑5.5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":85.6,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"terminal-bench-2-1-percent","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1-percent:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-330","ingestRunId":"launch-openai-sol-launch","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"Terminal-Bench","locator":"Coding table / Terminal-Bench 2.1 / Claude Fable 5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":83.1,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"terminal-bench-2-1-percent","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1-percent:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-331","ingestRunId":"launch-openai-sol-launch","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"Terminal-Bench","locator":"Coding table / Terminal-Bench 2.1 / Claude Opus 4.8","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":78.9,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"terminal-bench-2-1-percent","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1-percent:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-332","ingestRunId":"launch-openai-sol-launch","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"Terminal-Bench","locator":"Coding table / Terminal-Bench 2.1 / Gemini 3.1 Pro Preview","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":70.7,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"genebench-pro-not-specified","evidenceKind":"lab_self_report","harnessId":"genebench-pro-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-333","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"GeneBench Pro","locator":"Science And Health table / GeneBench Pro / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":28.7,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"genebench-pro-not-specified","evidenceKind":"lab_self_report","harnessId":"genebench-pro-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-334","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"GeneBench Pro","locator":"Science And Health table / GeneBench Pro / GPT‑5.6 Terra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":23.3,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"genebench-pro-not-specified","evidenceKind":"lab_self_report","harnessId":"genebench-pro-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-335","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-luna","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"GeneBench Pro","locator":"Science And Health table / GeneBench Pro / GPT‑5.6 Luna","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":10.8,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"genebench-pro-not-specified","evidenceKind":"lab_self_report","harnessId":"genebench-pro-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-336","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"GeneBench Pro","locator":"Science And Health table / GeneBench Pro / GPT‑5.5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":12,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"genebench-pro-not-specified","evidenceKind":"lab_self_report","harnessId":"genebench-pro-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-337","ingestRunId":"launch-openai-sol-launch","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"GeneBench Pro","locator":"Science And Health table / GeneBench Pro / Claude Opus 4.8","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":16,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"genebench-pro-not-specified","evidenceKind":"lab_self_report","harnessId":"genebench-pro-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-338","ingestRunId":"launch-openai-sol-launch","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"GeneBench Pro","locator":"Science And Health table / GeneBench Pro / Gemini 3.1 Pro Preview","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":3.1,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"genebench-pro-not-specified","evidenceKind":"lab_self_report","harnessId":"genebench-pro-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-339","ingestRunId":"launch-openai-sol-launch","modelId":"gemini-3.5-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"GeneBench Pro","locator":"Science And Health table / GeneBench Pro / Gemini 3.5 Flash","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":8.14,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"lifescibench-not-specified","evidenceKind":"lab_self_report","harnessId":"lifescibench-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-340","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"LifeSciBench","locator":"Science And Health table / LifeSciBench / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":59.9,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"lifescibench-not-specified","evidenceKind":"lab_self_report","harnessId":"lifescibench-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-341","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"LifeSciBench","locator":"Science And Health table / LifeSciBench / GPT‑5.6 Terra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":56,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"lifescibench-not-specified","evidenceKind":"lab_self_report","harnessId":"lifescibench-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-342","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-luna","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"LifeSciBench","locator":"Science And Health table / LifeSciBench / GPT‑5.6 Luna","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":51.2,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"lifescibench-not-specified","evidenceKind":"lab_self_report","harnessId":"lifescibench-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-343","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"LifeSciBench","locator":"Science And Health table / LifeSciBench / GPT‑5.5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":50.4,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"lifescibench-not-specified","evidenceKind":"lab_self_report","harnessId":"lifescibench-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-344","ingestRunId":"launch-openai-sol-launch","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"LifeSciBench","locator":"Science And Health table / LifeSciBench / Claude Opus 4.8","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":53.6,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"medchembench-internal-not-specified","evidenceKind":"lab_self_report","harnessId":"medchembench-internal-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-345","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"MedChemBench (Internal)","locator":"Science And Health table / MedChemBench (Internal) / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":48.3,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"medchembench-internal-not-specified","evidenceKind":"lab_self_report","harnessId":"medchembench-internal-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-346","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"MedChemBench (Internal)","locator":"Science And Health table / MedChemBench (Internal) / GPT‑5.6 Terra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":35,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"medchembench-internal-not-specified","evidenceKind":"lab_self_report","harnessId":"medchembench-internal-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-347","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-luna","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"MedChemBench (Internal)","locator":"Science And Health table / MedChemBench (Internal) / GPT‑5.6 Luna","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":30.4,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"medchembench-internal-not-specified","evidenceKind":"lab_self_report","harnessId":"medchembench-internal-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-348","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"MedChemBench (Internal)","locator":"Science And Health table / MedChemBench (Internal) / GPT‑5.5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":35.5,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"healthbench-professional-not-specified","evidenceKind":"lab_self_report","harnessId":"healthbench-professional-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IG9mZmljaWFsIHBhcGVyIHNjb3Jpbmc7IGxlbmd0aC1hZGp1c3RlZA","id":"launch-openai-sol-launch-349","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; official paper scoring; length-adjusted","family":"HealthBench Professional","locator":"Science And Health table / HealthBench Professional / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":60.5,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"healthbench-professional-not-specified","evidenceKind":"lab_self_report","harnessId":"healthbench-professional-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IG9mZmljaWFsIHBhcGVyIHNjb3Jpbmc7IGxlbmd0aC1hZGp1c3RlZA","id":"launch-openai-sol-launch-350","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; official paper scoring; length-adjusted","family":"HealthBench Professional","locator":"Science And Health table / HealthBench Professional / GPT‑5.6 Terra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":57.7,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"healthbench-professional-not-specified","evidenceKind":"lab_self_report","harnessId":"healthbench-professional-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IG9mZmljaWFsIHBhcGVyIHNjb3Jpbmc7IGxlbmd0aC1hZGp1c3RlZA","id":"launch-openai-sol-launch-351","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-luna","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; official paper scoring; length-adjusted","family":"HealthBench Professional","locator":"Science And Health table / HealthBench Professional / GPT‑5.6 Luna","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":55.7,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"healthbench-professional-not-specified","evidenceKind":"lab_self_report","harnessId":"healthbench-professional-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IG9mZmljaWFsIHBhcGVyIHNjb3Jpbmc7IGxlbmd0aC1hZGp1c3RlZA","id":"launch-openai-sol-launch-352","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; official paper scoring; length-adjusted","family":"HealthBench Professional","locator":"Science And Health table / HealthBench Professional / GPT‑5.5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":49.5,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"healthbench-professional-not-specified","evidenceKind":"lab_self_report","harnessId":"healthbench-professional-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IG9mZmljaWFsIHBhcGVyIHNjb3Jpbmc7IGxlbmd0aC1hZGp1c3RlZA","id":"launch-openai-sol-launch-353","ingestRunId":"launch-openai-sol-launch","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; official paper scoring; length-adjusted","family":"HealthBench Professional","locator":"Science And Health table / HealthBench Professional / Claude Fable 5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":60.9,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"healthbench-professional-not-specified","evidenceKind":"lab_self_report","harnessId":"healthbench-professional-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IG9mZmljaWFsIHBhcGVyIHNjb3Jpbmc7IGxlbmd0aC1hZGp1c3RlZA","id":"launch-openai-sol-launch-354","ingestRunId":"launch-openai-sol-launch","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"knowledge","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; official paper scoring; length-adjusted","family":"HealthBench Professional","locator":"Science And Health table / HealthBench Professional / Claude Opus 4.8","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":53,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"osworld-2-0-percent","evidenceKind":"lab_self_report","harnessId":"osworld-2-0-percent:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-355","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"OSWorld","locator":"Computer Use table / OSWorld 2.0 / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":62.6,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"osworld-2-0-percent","evidenceKind":"lab_self_report","harnessId":"osworld-2-0-percent:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-356","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"OSWorld","locator":"Computer Use table / OSWorld 2.0 / GPT‑5.6 Terra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":50.2,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"osworld-2-0-percent","evidenceKind":"lab_self_report","harnessId":"osworld-2-0-percent:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-357","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-luna","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"OSWorld","locator":"Computer Use table / OSWorld 2.0 / GPT‑5.6 Luna","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":45.6,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"osworld-2-0-percent","evidenceKind":"lab_self_report","harnessId":"osworld-2-0-percent:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-358","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"OSWorld","locator":"Computer Use table / OSWorld 2.0 / GPT‑5.5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":47.5,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"osworld-2-0-percent","evidenceKind":"lab_self_report","harnessId":"osworld-2-0-percent:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-359","ingestRunId":"launch-openai-sol-launch","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"OSWorld","locator":"Computer Use table / OSWorld 2.0 / Claude Opus 4.8","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":54.8,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"browsecomp-not-specified","evidenceKind":"lab_self_report","harnessId":"browsecomp-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-360","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"BrowseComp","locator":"Computer Use table / BrowseComp / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":90.4,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"browsecomp-not-specified","evidenceKind":"lab_self_report","harnessId":"browsecomp-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IFVsdHJhLCBmb3VyLWFnZW50IG9yY2hlc3RyYXRpb24","id":"launch-openai-sol-launch-361","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; Ultra, four-agent orchestration","family":"BrowseComp","locator":"Computer Use table / BrowseComp / GPT‑5.6 Sol Ultra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":92.2,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"browsecomp-not-specified","evidenceKind":"lab_self_report","harnessId":"browsecomp-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-362","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"BrowseComp","locator":"Computer Use table / BrowseComp / GPT‑5.6 Terra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":87.5,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"browsecomp-not-specified","evidenceKind":"lab_self_report","harnessId":"browsecomp-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-363","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-luna","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"BrowseComp","locator":"Computer Use table / BrowseComp / GPT‑5.6 Luna","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":83.3,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"browsecomp-not-specified","evidenceKind":"lab_self_report","harnessId":"browsecomp-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-364","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"BrowseComp","locator":"Computer Use table / BrowseComp / GPT‑5.5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":84.4,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"browsecomp-not-specified","evidenceKind":"lab_self_report","harnessId":"browsecomp-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-365","ingestRunId":"launch-openai-sol-launch","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"BrowseComp","locator":"Computer Use table / BrowseComp / Claude Opus 4.8","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":84.3,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"browsecomp-not-specified","evidenceKind":"lab_self_report","harnessId":"browsecomp-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-366","ingestRunId":"launch-openai-sol-launch","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"BrowseComp","locator":"Computer Use table / BrowseComp / Gemini 3.1 Pro Preview","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":85.9,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"benchcad-not-specified","evidenceKind":"lab_self_report","harnessId":"benchcad-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-367","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"BenchCAD","locator":"Computer Use table / BenchCAD / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":70.6,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"benchcad-not-specified","evidenceKind":"lab_self_report","harnessId":"benchcad-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-368","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"BenchCAD","locator":"Computer Use table / BenchCAD / GPT‑5.6 Terra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":62.3,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"benchcad-not-specified","evidenceKind":"lab_self_report","harnessId":"benchcad-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-369","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-luna","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"BenchCAD","locator":"Computer Use table / BenchCAD / GPT‑5.6 Luna","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":63.1,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"benchcad-not-specified","evidenceKind":"lab_self_report","harnessId":"benchcad-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-370","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"BenchCAD","locator":"Computer Use table / BenchCAD / GPT‑5.5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":44.4,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"benchcad-not-specified","evidenceKind":"lab_self_report","harnessId":"benchcad-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-371","ingestRunId":"launch-openai-sol-launch","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"BenchCAD","locator":"Computer Use table / BenchCAD / Claude Opus 4.8","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":27.3,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"benchcad-python-tool-not-specified","evidenceKind":"lab_self_report","harnessId":"benchcad-python-tool-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IFB5dGhvbiB0b29sIGVuYWJsZWQ","id":"launch-openai-sol-launch-372","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; Python tool enabled","family":"BenchCAD (python tool)","locator":"Computer Use table / BenchCAD (python tool) / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":83.4,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"benchcad-python-tool-not-specified","evidenceKind":"lab_self_report","harnessId":"benchcad-python-tool-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IFB5dGhvbiB0b29sIGVuYWJsZWQ","id":"launch-openai-sol-launch-373","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; Python tool enabled","family":"BenchCAD (python tool)","locator":"Computer Use table / BenchCAD (python tool) / GPT‑5.6 Terra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":78.2,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"benchcad-python-tool-not-specified","evidenceKind":"lab_self_report","harnessId":"benchcad-python-tool-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IFB5dGhvbiB0b29sIGVuYWJsZWQ","id":"launch-openai-sol-launch-374","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-luna","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; Python tool enabled","family":"BenchCAD (python tool)","locator":"Computer Use table / BenchCAD (python tool) / GPT‑5.6 Luna","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":73.9,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"benchcad-python-tool-not-specified","evidenceKind":"lab_self_report","harnessId":"benchcad-python-tool-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IFB5dGhvbiB0b29sIGVuYWJsZWQ","id":"launch-openai-sol-launch-375","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; Python tool enabled","family":"BenchCAD (python tool)","locator":"Computer Use table / BenchCAD (python tool) / GPT‑5.5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":55.8,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"benchcad-python-tool-not-specified","evidenceKind":"lab_self_report","harnessId":"benchcad-python-tool-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IFB5dGhvbiB0b29sIGVuYWJsZWQ","id":"launch-openai-sol-launch-376","ingestRunId":"launch-openai-sol-launch","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; Python tool enabled","family":"BenchCAD (python tool)","locator":"Computer Use table / BenchCAD (python tool) / Claude Opus 4.8","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":51.8,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"capture-the-flag-challenges-not-specified","evidenceKind":"lab_self_report","harnessId":"capture-the-flag-challenges-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IHJlZHVjZWQgb3IgYWJzZW50IHByb2R1Y3Rpb24gc2FmZWd1YXJkcw","id":"launch-openai-sol-launch-377","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; reduced or absent production safeguards","family":"Capture-the-Flag Challenges","locator":"Cybersecurity table / Capture-the-Flag Challenges / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":96.7,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"capture-the-flag-challenges-not-specified","evidenceKind":"lab_self_report","harnessId":"capture-the-flag-challenges-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IHJlZHVjZWQgb3IgYWJzZW50IHByb2R1Y3Rpb24gc2FmZWd1YXJkcw","id":"launch-openai-sol-launch-378","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; reduced or absent production safeguards","family":"Capture-the-Flag Challenges","locator":"Cybersecurity table / Capture-the-Flag Challenges / GPT‑5.6 Terra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":91.8,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"capture-the-flag-challenges-not-specified","evidenceKind":"lab_self_report","harnessId":"capture-the-flag-challenges-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IHJlZHVjZWQgb3IgYWJzZW50IHByb2R1Y3Rpb24gc2FmZWd1YXJkcw","id":"launch-openai-sol-launch-379","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-luna","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; reduced or absent production safeguards","family":"Capture-the-Flag Challenges","locator":"Cybersecurity table / Capture-the-Flag Challenges / GPT‑5.6 Luna","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":85.2,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"capture-the-flag-challenges-not-specified","evidenceKind":"lab_self_report","harnessId":"capture-the-flag-challenges-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IHJlZHVjZWQgb3IgYWJzZW50IHByb2R1Y3Rpb24gc2FmZWd1YXJkcw","id":"launch-openai-sol-launch-380","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; reduced or absent production safeguards","family":"Capture-the-Flag Challenges","locator":"Cybersecurity table / Capture-the-Flag Challenges / GPT‑5.5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":88.1,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"sec-bench-pro-may2026-public-grader","evidenceKind":"lab_self_report","harnessId":"sec-bench-pro-may2026-public-grader:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IHJlZHVjZWQgb3IgYWJzZW50IHByb2R1Y3Rpb24gc2FmZWd1YXJkczsgcHVibGljIGdyYWRlcjsgTWF5MjAyNiBKYXZhU2NyaXB0IHN1YnNldA","id":"launch-openai-sol-launch-381","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; reduced or absent production safeguards; public grader; May2026 JavaScript subset","family":"SEC-Bench Pro","locator":"Cybersecurity table / SEC-Bench Pro / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":71.2,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"sec-bench-pro-may2026-public-grader","evidenceKind":"lab_self_report","harnessId":"sec-bench-pro-may2026-public-grader:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IFVsdHJhLCBmb3VyLWFnZW50IG9yY2hlc3RyYXRpb247IHJlZHVjZWQgb3IgYWJzZW50IHByb2R1Y3Rpb24gc2FmZWd1YXJkczsgcHVibGljIGdyYWRlcjsgTWF5MjAyNiBKYXZhU2NyaXB0IHN1YnNldA","id":"launch-openai-sol-launch-382","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; Ultra, four-agent orchestration; reduced or absent production safeguards; public grader; May2026 JavaScript subset","family":"SEC-Bench Pro","locator":"Cybersecurity table / SEC-Bench Pro / GPT‑5.6 Sol Ultra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":74.3,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"sec-bench-pro-may2026-public-grader","evidenceKind":"lab_self_report","harnessId":"sec-bench-pro-may2026-public-grader:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IHJlZHVjZWQgb3IgYWJzZW50IHByb2R1Y3Rpb24gc2FmZWd1YXJkczsgcHVibGljIGdyYWRlcjsgTWF5MjAyNiBKYXZhU2NyaXB0IHN1YnNldA","id":"launch-openai-sol-launch-383","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; reduced or absent production safeguards; public grader; May2026 JavaScript subset","family":"SEC-Bench Pro","locator":"Cybersecurity table / SEC-Bench Pro / GPT‑5.6 Terra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":57.7,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"sec-bench-pro-may2026-public-grader","evidenceKind":"lab_self_report","harnessId":"sec-bench-pro-may2026-public-grader:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IHJlZHVjZWQgb3IgYWJzZW50IHByb2R1Y3Rpb24gc2FmZWd1YXJkczsgcHVibGljIGdyYWRlcjsgTWF5MjAyNiBKYXZhU2NyaXB0IHN1YnNldA","id":"launch-openai-sol-launch-384","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-luna","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; reduced or absent production safeguards; public grader; May2026 JavaScript subset","family":"SEC-Bench Pro","locator":"Cybersecurity table / SEC-Bench Pro / GPT‑5.6 Luna","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":48.9,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"sec-bench-pro-may2026-public-grader","evidenceKind":"lab_self_report","harnessId":"sec-bench-pro-may2026-public-grader:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IHJlZHVjZWQgb3IgYWJzZW50IHByb2R1Y3Rpb24gc2FmZWd1YXJkczsgcHVibGljIGdyYWRlcjsgTWF5MjAyNiBKYXZhU2NyaXB0IHN1YnNldA","id":"launch-openai-sol-launch-385","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; reduced or absent production safeguards; public grader; May2026 JavaScript subset","family":"SEC-Bench Pro","locator":"Cybersecurity table / SEC-Bench Pro / GPT‑5.5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":45.8,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"exploitbench-not-specified","evidenceKind":"lab_self_report","harnessId":"exploitbench-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IHJlZHVjZWQgb3IgYWJzZW50IHByb2R1Y3Rpb24gc2FmZWd1YXJkczsgRXhwbG9pdEJlbmNoIEFQSSBoYXJuZXNzOyBmaXZlIHNlZWRzOyByZWFzb25pbmcgY29udGludWl0eQ","id":"launch-openai-sol-launch-386","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; reduced or absent production safeguards; ExploitBench API harness; five seeds; reasoning continuity","family":"ExploitBench","locator":"Cybersecurity table / ExploitBench / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":73.5,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"exploitbench-not-specified","evidenceKind":"lab_self_report","harnessId":"exploitbench-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IHJlZHVjZWQgb3IgYWJzZW50IHByb2R1Y3Rpb24gc2FmZWd1YXJkczsgRXhwbG9pdEJlbmNoIEFQSSBoYXJuZXNzOyBmaXZlIHNlZWRzOyByZWFzb25pbmcgY29udGludWl0eQ","id":"launch-openai-sol-launch-387","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; reduced or absent production safeguards; ExploitBench API harness; five seeds; reasoning continuity","family":"ExploitBench","locator":"Cybersecurity table / ExploitBench / GPT‑5.6 Terra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":52.9,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"exploitbench-not-specified","evidenceKind":"lab_self_report","harnessId":"exploitbench-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IHJlZHVjZWQgb3IgYWJzZW50IHByb2R1Y3Rpb24gc2FmZWd1YXJkczsgRXhwbG9pdEJlbmNoIEFQSSBoYXJuZXNzOyBmaXZlIHNlZWRzOyByZWFzb25pbmcgY29udGludWl0eQ","id":"launch-openai-sol-launch-388","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-luna","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; reduced or absent production safeguards; ExploitBench API harness; five seeds; reasoning continuity","family":"ExploitBench","locator":"Cybersecurity table / ExploitBench / GPT‑5.6 Luna","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":33.2,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"exploitbench-not-specified","evidenceKind":"lab_self_report","harnessId":"exploitbench-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IHJlZHVjZWQgb3IgYWJzZW50IHByb2R1Y3Rpb24gc2FmZWd1YXJkczsgRXhwbG9pdEJlbmNoIEFQSSBoYXJuZXNzOyBmaXZlIHNlZWRzOyByZWFzb25pbmcgY29udGludWl0eQ","id":"launch-openai-sol-launch-389","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; reduced or absent production safeguards; ExploitBench API harness; five seeds; reasoning continuity","family":"ExploitBench","locator":"Cybersecurity table / ExploitBench / GPT‑5.5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":47.9,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"exploitbench-not-specified","evidenceKind":"lab_self_report","harnessId":"exploitbench-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IHJlZHVjZWQgb3IgYWJzZW50IHByb2R1Y3Rpb24gc2FmZWd1YXJkczsgRXhwbG9pdEJlbmNoIEFQSSBoYXJuZXNzOyBmaXZlIHNlZWRzOyByZWFzb25pbmcgY29udGludWl0eQ","id":"launch-openai-sol-launch-390","ingestRunId":"launch-openai-sol-launch","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; reduced or absent production safeguards; ExploitBench API harness; five seeds; reasoning continuity","family":"ExploitBench","locator":"Cybersecurity table / ExploitBench / Claude Opus 4.8","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":40,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"exploitgym-not-specified","evidenceKind":"lab_self_report","harnessId":"exploitgym-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IHJlZHVjZWQgb3IgYWJzZW50IHByb2R1Y3Rpb24gc2FmZWd1YXJkczsgc2l4LWhvdXIgZXZhbHVhdGlvbiBjYXA7IGFscGhhIEFQSSBsYXRlbmN5IHJlc2NhbGVkIHRvIHB1YmxpYyBBUEk","id":"launch-openai-sol-launch-391","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; reduced or absent production safeguards; six-hour evaluation cap; alpha API latency rescaled to public API","family":"ExploitGym","locator":"Cybersecurity table / ExploitGym / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":33.7,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"exploitgym-not-specified","evidenceKind":"lab_self_report","harnessId":"exploitgym-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IHJlZHVjZWQgb3IgYWJzZW50IHByb2R1Y3Rpb24gc2FmZWd1YXJkczsgc2l4LWhvdXIgZXZhbHVhdGlvbiBjYXA7IGFscGhhIEFQSSBsYXRlbmN5IHJlc2NhbGVkIHRvIHB1YmxpYyBBUEk","id":"launch-openai-sol-launch-392","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; reduced or absent production safeguards; six-hour evaluation cap; alpha API latency rescaled to public API","family":"ExploitGym","locator":"Cybersecurity table / ExploitGym / GPT‑5.6 Terra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":23.2,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"exploitgym-not-specified","evidenceKind":"lab_self_report","harnessId":"exploitgym-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IHJlZHVjZWQgb3IgYWJzZW50IHByb2R1Y3Rpb24gc2FmZWd1YXJkczsgc2l4LWhvdXIgZXZhbHVhdGlvbiBjYXA7IGFscGhhIEFQSSBsYXRlbmN5IHJlc2NhbGVkIHRvIHB1YmxpYyBBUEk","id":"launch-openai-sol-launch-393","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-luna","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; reduced or absent production safeguards; six-hour evaluation cap; alpha API latency rescaled to public API","family":"ExploitGym","locator":"Cybersecurity table / ExploitGym / GPT‑5.6 Luna","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":12.4,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"exploitgym-not-specified","evidenceKind":"lab_self_report","harnessId":"exploitgym-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IHJlZHVjZWQgb3IgYWJzZW50IHByb2R1Y3Rpb24gc2FmZWd1YXJkczsgc2l4LWhvdXIgZXZhbHVhdGlvbiBjYXA7IGFscGhhIEFQSSBsYXRlbmN5IHJlc2NhbGVkIHRvIHB1YmxpYyBBUEk","id":"launch-openai-sol-launch-394","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; reduced or absent production safeguards; six-hour evaluation cap; alpha API latency rescaled to public API","family":"ExploitGym","locator":"Cybersecurity table / ExploitGym / GPT‑5.5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":15.1,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"internal-research-debugging-evaluation-not-specified","evidenceKind":"lab_self_report","harnessId":"internal-research-debugging-evaluation-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-395","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"Internal Research Debugging Evaluation","locator":"Self-Improvement table / Internal Research Debugging Evaluation / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":68.3,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"internal-research-debugging-evaluation-not-specified","evidenceKind":"lab_self_report","harnessId":"internal-research-debugging-evaluation-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-396","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"Internal Research Debugging Evaluation","locator":"Self-Improvement table / Internal Research Debugging Evaluation / GPT‑5.6 Terra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":67.8,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"internal-research-debugging-evaluation-not-specified","evidenceKind":"lab_self_report","harnessId":"internal-research-debugging-evaluation-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-397","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-luna","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"Internal Research Debugging Evaluation","locator":"Self-Improvement table / Internal Research Debugging Evaluation / GPT‑5.6 Luna","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":50.8,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"internal-research-debugging-evaluation-not-specified","evidenceKind":"lab_self_report","harnessId":"internal-research-debugging-evaluation-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-398","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"Internal Research Debugging Evaluation","locator":"Self-Improvement table / Internal Research Debugging Evaluation / GPT‑5.5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":50,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"kernelgen-1p-not-specified","evidenceKind":"lab_self_report","harnessId":"kernelgen-1p-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-399","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"KernelGen 1P","locator":"Self-Improvement table / KernelGen 1P / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":61.1,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"kernelgen-1p-not-specified","evidenceKind":"lab_self_report","harnessId":"kernelgen-1p-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-400","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"KernelGen 1P","locator":"Self-Improvement table / KernelGen 1P / GPT‑5.6 Terra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":49.2,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"kernelgen-1p-not-specified","evidenceKind":"lab_self_report","harnessId":"kernelgen-1p-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-401","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-luna","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"KernelGen 1P","locator":"Self-Improvement table / KernelGen 1P / GPT‑5.6 Luna","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":22.4,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"kernelgen-1p-not-specified","evidenceKind":"lab_self_report","harnessId":"kernelgen-1p-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-402","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"KernelGen 1P","locator":"Self-Improvement table / KernelGen 1P / GPT‑5.5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":29.3,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"nanogpt-not-specified","evidenceKind":"lab_self_report","harnessId":"nanogpt-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-403","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"NanoGPT","locator":"Self-Improvement table / NanoGPT / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":9.69,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"nanogpt-not-specified","evidenceKind":"lab_self_report","harnessId":"nanogpt-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-404","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"NanoGPT","locator":"Self-Improvement table / NanoGPT / GPT‑5.6 Terra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":14.5,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"nanogpt-not-specified","evidenceKind":"lab_self_report","harnessId":"nanogpt-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-405","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-luna","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"NanoGPT","locator":"Self-Improvement table / NanoGPT / GPT‑5.6 Luna","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":1.66,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"nanogpt-not-specified","evidenceKind":"lab_self_report","harnessId":"nanogpt-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-406","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"NanoGPT","locator":"Self-Improvement table / NanoGPT / GPT‑5.5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":2.65,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"posttrainbench-lite-not-specified","evidenceKind":"lab_self_report","harnessId":"posttrainbench-lite-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-407","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"PostTrainBench Lite","locator":"Self-Improvement table / PostTrainBench Lite / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":50.3,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"posttrainbench-lite-not-specified","evidenceKind":"lab_self_report","harnessId":"posttrainbench-lite-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-408","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"PostTrainBench Lite","locator":"Self-Improvement table / PostTrainBench Lite / GPT‑5.6 Terra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":51.5,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"posttrainbench-lite-not-specified","evidenceKind":"lab_self_report","harnessId":"posttrainbench-lite-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-409","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-luna","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"PostTrainBench Lite","locator":"Self-Improvement table / PostTrainBench Lite / GPT‑5.6 Luna","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":29.6,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"posttrainbench-lite-not-specified","evidenceKind":"lab_self_report","harnessId":"posttrainbench-lite-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-410","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"PostTrainBench Lite","locator":"Self-Improvement table / PostTrainBench Lite / GPT‑5.5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":38.8,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"rsi-index-not-specified","evidenceKind":"lab_self_report","harnessId":"rsi-index-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-411","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"RSI Index","locator":"Self-Improvement table / RSI Index / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":57.9,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"rsi-index-not-specified","evidenceKind":"lab_self_report","harnessId":"rsi-index-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-412","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"RSI Index","locator":"Self-Improvement table / RSI Index / GPT‑5.6 Terra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":56.3,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"rsi-index-not-specified","evidenceKind":"lab_self_report","harnessId":"rsi-index-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-413","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-luna","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"RSI Index","locator":"Self-Improvement table / RSI Index / GPT‑5.6 Luna","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":41.9,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"rsi-index-not-specified","evidenceKind":"lab_self_report","harnessId":"rsi-index-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-414","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"RSI Index","locator":"Self-Improvement table / RSI Index / GPT‑5.5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":41.7,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"mmmu-pro-no-tools-not-specified","evidenceKind":"lab_self_report","harnessId":"mmmu-pro-no-tools-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IG5vIHRvb2xz","id":"launch-openai-sol-launch-415","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; no tools","family":"MMMU Pro (no tools)","locator":"Multimodal table / MMMU Pro (no tools) / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":83,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"mmmu-pro-no-tools-not-specified","evidenceKind":"lab_self_report","harnessId":"mmmu-pro-no-tools-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IG5vIHRvb2xz","id":"launch-openai-sol-launch-416","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; no tools","family":"MMMU Pro (no tools)","locator":"Multimodal table / MMMU Pro (no tools) / GPT‑5.6 Terra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":80.7,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"mmmu-pro-no-tools-not-specified","evidenceKind":"lab_self_report","harnessId":"mmmu-pro-no-tools-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IG5vIHRvb2xz","id":"launch-openai-sol-launch-417","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-luna","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; no tools","family":"MMMU Pro (no tools)","locator":"Multimodal table / MMMU Pro (no tools) / GPT‑5.6 Luna","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":78.4,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"mmmu-pro-no-tools-not-specified","evidenceKind":"lab_self_report","harnessId":"mmmu-pro-no-tools-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IG5vIHRvb2xz","id":"launch-openai-sol-launch-418","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; no tools","family":"MMMU Pro (no tools)","locator":"Multimodal table / MMMU Pro (no tools) / GPT‑5.5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":81.2,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"mmmu-pro-no-tools-not-specified","evidenceKind":"lab_self_report","harnessId":"mmmu-pro-no-tools-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IG5vIHRvb2xz","id":"launch-openai-sol-launch-419","ingestRunId":"launch-openai-sol-launch","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; no tools","family":"MMMU Pro (no tools)","locator":"Multimodal table / MMMU Pro (no tools) / Gemini 3.1 Pro Preview","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":80.5,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"mmmu-pro-with-tools-not-specified","evidenceKind":"lab_self_report","harnessId":"mmmu-pro-with-tools-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IHRvb2xzIGVuYWJsZWQ","id":"launch-openai-sol-launch-420","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; tools enabled","family":"MMMU Pro (with tools)","locator":"Multimodal table / MMMU Pro (with tools) / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":84.6,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"mmmu-pro-with-tools-not-specified","evidenceKind":"lab_self_report","harnessId":"mmmu-pro-with-tools-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IHRvb2xzIGVuYWJsZWQ","id":"launch-openai-sol-launch-421","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; tools enabled","family":"MMMU Pro (with tools)","locator":"Multimodal table / MMMU Pro (with tools) / GPT‑5.6 Terra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":82,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"mmmu-pro-with-tools-not-specified","evidenceKind":"lab_self_report","harnessId":"mmmu-pro-with-tools-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IHRvb2xzIGVuYWJsZWQ","id":"launch-openai-sol-launch-422","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-luna","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; tools enabled","family":"MMMU Pro (with tools)","locator":"Multimodal table / MMMU Pro (with tools) / GPT‑5.6 Luna","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":79.5,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"mmmu-pro-with-tools-not-specified","evidenceKind":"lab_self_report","harnessId":"mmmu-pro-with-tools-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IHRvb2xzIGVuYWJsZWQ","id":"launch-openai-sol-launch-423","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; tools enabled","family":"MMMU Pro (with tools)","locator":"Multimodal table / MMMU Pro (with tools) / GPT‑5.5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":83.2,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"gdp-pdf-not-specified","evidenceKind":"lab_self_report","harnessId":"gdp-pdf-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-424","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"gdp.pdf","locator":"Multimodal table / gdp.pdf / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":30.7,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"gdp-pdf-not-specified","evidenceKind":"lab_self_report","harnessId":"gdp-pdf-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-425","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"gdp.pdf","locator":"Multimodal table / gdp.pdf / GPT‑5.6 Terra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":24.7,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"gdp-pdf-not-specified","evidenceKind":"lab_self_report","harnessId":"gdp-pdf-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-426","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-luna","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"gdp.pdf","locator":"Multimodal table / gdp.pdf / GPT‑5.6 Luna","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":22.7,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"gdp-pdf-not-specified","evidenceKind":"lab_self_report","harnessId":"gdp-pdf-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-427","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"gdp.pdf","locator":"Multimodal table / gdp.pdf / GPT‑5.5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":26,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"gdp-pdf-not-specified","evidenceKind":"lab_self_report","harnessId":"gdp-pdf-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-428","ingestRunId":"launch-openai-sol-launch","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"gdp.pdf","locator":"Multimodal table / gdp.pdf / Claude Fable 5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":29.8,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"gdp-pdf-not-specified","evidenceKind":"lab_self_report","harnessId":"gdp-pdf-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-429","ingestRunId":"launch-openai-sol-launch","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"gdp.pdf","locator":"Multimodal table / gdp.pdf / Claude Opus 4.8","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":22.5,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"gdp-pdf-not-specified","evidenceKind":"lab_self_report","harnessId":"gdp-pdf-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-430","ingestRunId":"launch-openai-sol-launch","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"multimodal","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"gdp.pdf","locator":"Multimodal table / gdp.pdf / Gemini 3.1 Pro Preview","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":16.7,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"gpqa-diamond-not-specified","evidenceKind":"lab_self_report","harnessId":"gpqa-diamond-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-431","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"GPQA Diamond","locator":"Academic table / GPQA Diamond / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":94.6,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"gpqa-diamond-not-specified","evidenceKind":"lab_self_report","harnessId":"gpqa-diamond-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-432","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"GPQA Diamond","locator":"Academic table / GPQA Diamond / GPT‑5.6 Terra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":92.9,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"gpqa-diamond-not-specified","evidenceKind":"lab_self_report","harnessId":"gpqa-diamond-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-433","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-luna","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"GPQA Diamond","locator":"Academic table / GPQA Diamond / GPT‑5.6 Luna","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":92.3,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"gpqa-diamond-not-specified","evidenceKind":"lab_self_report","harnessId":"gpqa-diamond-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-434","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"GPQA Diamond","locator":"Academic table / GPQA Diamond / GPT‑5.5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":93.6,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"gpqa-diamond-not-specified","evidenceKind":"lab_self_report","harnessId":"gpqa-diamond-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-435","ingestRunId":"launch-openai-sol-launch","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"GPQA Diamond","locator":"Academic table / GPQA Diamond / Claude Fable 5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":92.6,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"gpqa-diamond-not-specified","evidenceKind":"lab_self_report","harnessId":"gpqa-diamond-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-436","ingestRunId":"launch-openai-sol-launch","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"GPQA Diamond","locator":"Academic table / GPQA Diamond / Claude Opus 4.8","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":92,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"gpqa-diamond-not-specified","evidenceKind":"lab_self_report","harnessId":"gpqa-diamond-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-437","ingestRunId":"launch-openai-sol-launch","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"GPQA Diamond","locator":"Academic table / GPQA Diamond / Gemini 3.1 Pro Preview","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":94.3,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"frontiermath-tier-1-3-2","evidenceKind":"lab_self_report","harnessId":"frontiermath-tier-1-3-2:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-438","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"FrontierMath Tier 1-3 (v2)","locator":"Academic table / FrontierMath Tier 1-3 (v2) / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":89,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"frontiermath-tier-1-3-2","evidenceKind":"lab_self_report","harnessId":"frontiermath-tier-1-3-2:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-439","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"FrontierMath Tier 1-3 (v2)","locator":"Academic table / FrontierMath Tier 1-3 (v2) / GPT‑5.6 Terra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":84.9,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"frontiermath-tier-1-3-2","evidenceKind":"lab_self_report","harnessId":"frontiermath-tier-1-3-2:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-440","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-luna","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"FrontierMath Tier 1-3 (v2)","locator":"Academic table / FrontierMath Tier 1-3 (v2) / GPT‑5.6 Luna","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":78.6,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"frontiermath-tier-1-3-2","evidenceKind":"lab_self_report","harnessId":"frontiermath-tier-1-3-2:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-441","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"FrontierMath Tier 1-3 (v2)","locator":"Academic table / FrontierMath Tier 1-3 (v2) / GPT‑5.5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":85.3,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"frontiermath-tier-1-3-2","evidenceKind":"lab_self_report","harnessId":"frontiermath-tier-1-3-2:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-442","ingestRunId":"launch-openai-sol-launch","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"FrontierMath Tier 1-3 (v2)","locator":"Academic table / FrontierMath Tier 1-3 (v2) / Claude Fable 5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":87,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"frontiermath-tier-1-3-2","evidenceKind":"lab_self_report","harnessId":"frontiermath-tier-1-3-2:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-443","ingestRunId":"launch-openai-sol-launch","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"FrontierMath Tier 1-3 (v2)","locator":"Academic table / FrontierMath Tier 1-3 (v2) / Claude Opus 4.8","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":80,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"frontiermath-tier-1-3-2","evidenceKind":"lab_self_report","harnessId":"frontiermath-tier-1-3-2:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-444","ingestRunId":"launch-openai-sol-launch","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"FrontierMath Tier 1-3 (v2)","locator":"Academic table / FrontierMath Tier 1-3 (v2) / Gemini 3.1 Pro Preview","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":59.6,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"frontiermath-tier-4-2","evidenceKind":"lab_self_report","harnessId":"frontiermath-tier-4-2:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-445","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"FrontierMath Tier 4 (v2)","locator":"Academic table / FrontierMath Tier 4 (v2) / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":83,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"frontiermath-tier-4-2","evidenceKind":"lab_self_report","harnessId":"frontiermath-tier-4-2:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-446","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"FrontierMath Tier 4 (v2)","locator":"Academic table / FrontierMath Tier 4 (v2) / GPT‑5.6 Terra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":68.3,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"frontiermath-tier-4-2","evidenceKind":"lab_self_report","harnessId":"frontiermath-tier-4-2:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-447","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-luna","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"FrontierMath Tier 4 (v2)","locator":"Academic table / FrontierMath Tier 4 (v2) / GPT‑5.6 Luna","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":58.5,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"frontiermath-tier-4-2","evidenceKind":"lab_self_report","harnessId":"frontiermath-tier-4-2:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-448","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"FrontierMath Tier 4 (v2)","locator":"Academic table / FrontierMath Tier 4 (v2) / GPT‑5.5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":72.5,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"frontiermath-tier-4-2","evidenceKind":"lab_self_report","harnessId":"frontiermath-tier-4-2:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-449","ingestRunId":"launch-openai-sol-launch","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"FrontierMath Tier 4 (v2)","locator":"Academic table / FrontierMath Tier 4 (v2) / Claude Fable 5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":87.8,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"frontiermath-tier-4-2","evidenceKind":"lab_self_report","harnessId":"frontiermath-tier-4-2:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-450","ingestRunId":"launch-openai-sol-launch","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"FrontierMath Tier 4 (v2)","locator":"Academic table / FrontierMath Tier 4 (v2) / Claude Opus 4.8","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":56.1,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"automationbench-not-specified","evidenceKind":"lab_self_report","harnessId":"automationbench-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-451","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"AutomationBench","locator":"Tool Use table / AutomationBench / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":18.1,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"automationbench-not-specified","evidenceKind":"lab_self_report","harnessId":"automationbench-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-452","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"AutomationBench","locator":"Tool Use table / AutomationBench / GPT‑5.6 Terra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":15.2,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"automationbench-not-specified","evidenceKind":"lab_self_report","harnessId":"automationbench-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-453","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-luna","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"AutomationBench","locator":"Tool Use table / AutomationBench / GPT‑5.6 Luna","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":14.9,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"automationbench-not-specified","evidenceKind":"lab_self_report","harnessId":"automationbench-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-454","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"AutomationBench","locator":"Tool Use table / AutomationBench / GPT‑5.5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":12.9,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"automationbench-not-specified","evidenceKind":"lab_self_report","harnessId":"automationbench-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-455","ingestRunId":"launch-openai-sol-launch","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"AutomationBench","locator":"Tool Use table / AutomationBench / Claude Fable 5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":17.4,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"automationbench-not-specified","evidenceKind":"lab_self_report","harnessId":"automationbench-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-456","ingestRunId":"launch-openai-sol-launch","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"AutomationBench","locator":"Tool Use table / AutomationBench / Claude Opus 4.8","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":15.5,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"automationbench-not-specified","evidenceKind":"lab_self_report","harnessId":"automationbench-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-457","ingestRunId":"launch-openai-sol-launch","modelId":"gemini-3.5-flash","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"AutomationBench","locator":"Tool Use table / AutomationBench / Gemini 3.5 Flash","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":14.5,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"toolathlon-not-specified","evidenceKind":"lab_self_report","harnessId":"toolathlon-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-458","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"Toolathlon","locator":"Tool Use table / Toolathlon / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":58,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"toolathlon-not-specified","evidenceKind":"lab_self_report","harnessId":"toolathlon-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-459","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"Toolathlon","locator":"Tool Use table / Toolathlon / GPT‑5.6 Terra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":53.1,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"toolathlon-not-specified","evidenceKind":"lab_self_report","harnessId":"toolathlon-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-460","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-luna","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"Toolathlon","locator":"Tool Use table / Toolathlon / GPT‑5.6 Luna","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":53.4,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"toolathlon-not-specified","evidenceKind":"lab_self_report","harnessId":"toolathlon-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-461","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"Toolathlon","locator":"Tool Use table / Toolathlon / GPT‑5.5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":55.6,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"toolathlon-not-specified","evidenceKind":"lab_self_report","harnessId":"toolathlon-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-462","ingestRunId":"launch-openai-sol-launch","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"Toolathlon","locator":"Tool Use table / Toolathlon / Claude Fable 5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":61.7,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"toolathlon-not-specified","evidenceKind":"lab_self_report","harnessId":"toolathlon-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-463","ingestRunId":"launch-openai-sol-launch","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"Toolathlon","locator":"Tool Use table / Toolathlon / Claude Opus 4.8","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":59.9,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"toolathlon-not-specified","evidenceKind":"lab_self_report","harnessId":"toolathlon-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-464","ingestRunId":"launch-openai-sol-launch","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"Toolathlon","locator":"Tool Use table / Toolathlon / Gemini 3.1 Pro Preview","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":48.8,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"openai-mrcr-v2-8-needle-256k-512k-2","evidenceKind":"lab_self_report","harnessId":"openai-mrcr-v2-8-needle-256k-512k-2:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IDgtbmVlZGxlIDI1NkstNTEySw","id":"launch-openai-sol-launch-465","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; 8-needle 256K-512K","family":"OpenAI MRCR v2 8-needle 256K-512K","locator":"Long Context table / OpenAI MRCR v2 8-needle 256K-512K / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":91.5,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"openai-mrcr-v2-8-needle-256k-512k-2","evidenceKind":"lab_self_report","harnessId":"openai-mrcr-v2-8-needle-256k-512k-2:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IDgtbmVlZGxlIDI1NkstNTEySw","id":"launch-openai-sol-launch-466","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; 8-needle 256K-512K","family":"OpenAI MRCR v2 8-needle 256K-512K","locator":"Long Context table / OpenAI MRCR v2 8-needle 256K-512K / GPT‑5.6 Terra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":89.6,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"openai-mrcr-v2-8-needle-256k-512k-2","evidenceKind":"lab_self_report","harnessId":"openai-mrcr-v2-8-needle-256k-512k-2:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IDgtbmVlZGxlIDI1NkstNTEySw","id":"launch-openai-sol-launch-467","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-luna","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; 8-needle 256K-512K","family":"OpenAI MRCR v2 8-needle 256K-512K","locator":"Long Context table / OpenAI MRCR v2 8-needle 256K-512K / GPT‑5.6 Luna","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":41.3,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"openai-mrcr-v2-8-needle-256k-512k-2","evidenceKind":"lab_self_report","harnessId":"openai-mrcr-v2-8-needle-256k-512k-2:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IDgtbmVlZGxlIDI1NkstNTEySw","id":"launch-openai-sol-launch-468","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; 8-needle 256K-512K","family":"OpenAI MRCR v2 8-needle 256K-512K","locator":"Long Context table / OpenAI MRCR v2 8-needle 256K-512K / GPT‑5.5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":81.5,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"openai-mrcr-v2-8-needle-512k-1m-2","evidenceKind":"lab_self_report","harnessId":"openai-mrcr-v2-8-needle-512k-1m-2:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IDgtbmVlZGxlIDUxMkstMU0","id":"launch-openai-sol-launch-469","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; 8-needle 512K-1M","family":"OpenAI MRCR v2 8-needle 512K-1M","locator":"Long Context table / OpenAI MRCR v2 8-needle 512K-1M / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":73.8,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"openai-mrcr-v2-8-needle-512k-1m-2","evidenceKind":"lab_self_report","harnessId":"openai-mrcr-v2-8-needle-512k-1m-2:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IDgtbmVlZGxlIDUxMkstMU0","id":"launch-openai-sol-launch-470","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; 8-needle 512K-1M","family":"OpenAI MRCR v2 8-needle 512K-1M","locator":"Long Context table / OpenAI MRCR v2 8-needle 512K-1M / GPT‑5.6 Terra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":72.5,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"openai-mrcr-v2-8-needle-512k-1m-2","evidenceKind":"lab_self_report","harnessId":"openai-mrcr-v2-8-needle-512k-1m-2:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IDgtbmVlZGxlIDUxMkstMU0","id":"launch-openai-sol-launch-471","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-luna","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; 8-needle 512K-1M","family":"OpenAI MRCR v2 8-needle 512K-1M","locator":"Long Context table / OpenAI MRCR v2 8-needle 512K-1M / GPT‑5.6 Luna","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":41.3,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"openai-mrcr-v2-8-needle-512k-1m-2","evidenceKind":"lab_self_report","harnessId":"openai-mrcr-v2-8-needle-512k-1m-2:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IDgtbmVlZGxlIDUxMkstMU0","id":"launch-openai-sol-launch-472","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; 8-needle 512K-1M","family":"OpenAI MRCR v2 8-needle 512K-1M","locator":"Long Context table / OpenAI MRCR v2 8-needle 512K-1M / GPT‑5.5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":74,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"graphwalks-bfs-256k-f1-not-specified","evidenceKind":"lab_self_report","harnessId":"graphwalks-bfs-256k-f1-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IEJGUyAyNTZrIGYx","id":"launch-openai-sol-launch-473","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; BFS 256k f1","family":"GraphWalks BFS 256k f1","locator":"Long Context table / GraphWalks BFS 256k f1 / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":90.7,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"graphwalks-bfs-256k-f1-not-specified","evidenceKind":"lab_self_report","harnessId":"graphwalks-bfs-256k-f1-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IEJGUyAyNTZrIGYx","id":"launch-openai-sol-launch-474","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; BFS 256k f1","family":"GraphWalks BFS 256k f1","locator":"Long Context table / GraphWalks BFS 256k f1 / GPT‑5.6 Terra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":76.9,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"graphwalks-bfs-256k-f1-not-specified","evidenceKind":"lab_self_report","harnessId":"graphwalks-bfs-256k-f1-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IEJGUyAyNTZrIGYx","id":"launch-openai-sol-launch-475","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-luna","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; BFS 256k f1","family":"GraphWalks BFS 256k f1","locator":"Long Context table / GraphWalks BFS 256k f1 / GPT‑5.6 Luna","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":81.3,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"graphwalks-bfs-256k-f1-not-specified","evidenceKind":"lab_self_report","harnessId":"graphwalks-bfs-256k-f1-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IEJGUyAyNTZrIGYx","id":"launch-openai-sol-launch-476","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; BFS 256k f1","family":"GraphWalks BFS 256k f1","locator":"Long Context table / GraphWalks BFS 256k f1 / GPT‑5.5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":73.7,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"graphwalks-bfs-256k-f1-not-specified","evidenceKind":"lab_self_report","harnessId":"graphwalks-bfs-256k-f1-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IEJGUyAyNTZrIGYx","id":"launch-openai-sol-launch-477","ingestRunId":"launch-openai-sol-launch","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; BFS 256k f1","family":"GraphWalks BFS 256k f1","locator":"Long Context table / GraphWalks BFS 256k f1 / Claude Opus 4.8","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":85.9,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"graphwalks-bfs-1mil-f1-not-specified","evidenceKind":"lab_self_report","harnessId":"graphwalks-bfs-1mil-f1-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IEJGUyAxbWlsIGYx","id":"launch-openai-sol-launch-478","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; BFS 1mil f1","family":"GraphWalks BFS 1mil f1","locator":"Long Context table / GraphWalks BFS 1mil f1 / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":77.1,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"graphwalks-bfs-1mil-f1-not-specified","evidenceKind":"lab_self_report","harnessId":"graphwalks-bfs-1mil-f1-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IEJGUyAxbWlsIGYx","id":"launch-openai-sol-launch-479","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; BFS 1mil f1","family":"GraphWalks BFS 1mil f1","locator":"Long Context table / GraphWalks BFS 1mil f1 / GPT‑5.6 Terra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":71.2,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"graphwalks-bfs-1mil-f1-not-specified","evidenceKind":"lab_self_report","harnessId":"graphwalks-bfs-1mil-f1-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IEJGUyAxbWlsIGYx","id":"launch-openai-sol-launch-480","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-luna","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; BFS 1mil f1","family":"GraphWalks BFS 1mil f1","locator":"Long Context table / GraphWalks BFS 1mil f1 / GPT‑5.6 Luna","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":51.2,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"graphwalks-bfs-1mil-f1-not-specified","evidenceKind":"lab_self_report","harnessId":"graphwalks-bfs-1mil-f1-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IEJGUyAxbWlsIGYx","id":"launch-openai-sol-launch-481","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; BFS 1mil f1","family":"GraphWalks BFS 1mil f1","locator":"Long Context table / GraphWalks BFS 1mil f1 / GPT‑5.5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":45.4,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"graphwalks-bfs-1mil-f1-not-specified","evidenceKind":"lab_self_report","harnessId":"graphwalks-bfs-1mil-f1-not-specified:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IEJGUyAxbWlsIGYx","id":"launch-openai-sol-launch-482","ingestRunId":"launch-openai-sol-launch","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"long_context","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; BFS 1mil f1","family":"GraphWalks BFS 1mil f1","locator":"Long Context table / GraphWalks BFS 1mil f1 / Claude Opus 4.8","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":68.1,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"arc-agi-3","evidenceKind":"lab_self_report","harnessId":"arc-agi-3:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-483","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"ARC-AGI-3","locator":"Abstract Reasoning table / ARC-AGI-3 / GPT‑5.6 Sol","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":7.78,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"arc-agi-3","evidenceKind":"lab_self_report","harnessId":"arc-agi-3:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-484","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-terra","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"ARC-AGI-3","locator":"Abstract Reasoning table / ARC-AGI-3 / GPT‑5.6 Terra","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":0.8,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"arc-agi-3","evidenceKind":"lab_self_report","harnessId":"arc-agi-3:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-485","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-luna","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"ARC-AGI-3","locator":"Abstract Reasoning table / ARC-AGI-3 / GPT‑5.6 Luna","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":0.18,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"arc-agi-3","evidenceKind":"lab_self_report","harnessId":"arc-agi-3:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-486","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"ARC-AGI-3","locator":"Abstract Reasoning table / ARC-AGI-3 / GPT‑5.5","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":0.43,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"arc-agi-3","evidenceKind":"lab_self_report","harnessId":"arc-agi-3:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ7IGhpZ2ggcmVhc29uaW5nLCBub3QgbWF4","id":"launch-openai-sol-launch-487","ingestRunId":"launch-openai-sol-launch","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; high reasoning, not max","family":"ARC-AGI-3","locator":"Abstract Reasoning table / ARC-AGI-3 / Claude Opus 4.8","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":1.5,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"arc-agi-3","evidenceKind":"lab_self_report","harnessId":"arc-agi-3:openai-sol-launch:TGF1bmNoIHRhYmxlIHJlcG9ydGVkIGNvbmZpZ3VyYXRpb247IHBlci1jZWxsIHJlYXNvbmluZyBlZmZvcnQgdW5zcGVjaWZpZWQ","id":"launch-openai-sol-launch-488","ingestRunId":"launch-openai-sol-launch","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"hard_reasoning","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified","family":"ARC-AGI-3","locator":"Abstract Reasoning table / ARC-AGI-3 / Gemini 3.1 Pro Preview","metric":"percent","notes":"Provider-published result; comparator measurements are not automatically independently reproduced.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":0.42,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"exploitgym-not-specified","evidenceKind":"lab_self_report","harnessId":"exploitgym-not-specified:openai-sol-launch:VHdvLWhvdXIgY2FwOyBhbHBoYSBBUEkgbGF0ZW5jeSByZXNjYWxlZDsgcmVkdWNlZCBzYWZlZ3VhcmRz","id":"launch-openai-sol-launch-550","ingestRunId":"launch-openai-sol-launch","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Two-hour cap; alpha API latency rescaled; reduced safeguards","family":"ExploitGym","locator":"Pushing the frontier on cyber and science paragraph","metric":"percent","notes":"Provider-published result; preserves source setting without claiming a controlled cross-provider comparison.","sourceId":"openai-sol-launch","sourceTitle":"GPT-5.6: Frontier intelligence that scales with your ambition"},"score":24.9,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"terminal-bench-2-1","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gUXdlbiBDbGF1ZGUgQ29kZSBhdmdAMTAsNWggdGltZW91dCwxMzEwNzIgb3V0cHV0OyBjb21wYXJhdG9ycyBiZXN0IHB1Ymxpc2hlZCBhY3Jvc3MgaGFybmVzc2VzLg","id":"launch-qwen-551","ingestRunId":"launch-qwen","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"coding","comparabilityReason":"The Qwen card cites an external leaderboard or release report for this comparator; it is not a new Qwen evaluation.","comparable":false,"configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Qwen Claude Code avg@10,5h timeout,131072 output; comparators best published across harnesses.","family":"Terminal-Bench","locator":"Performance table: Terminal Bench 2.1","metric":"%","notes":"Cited comparator result; see the benchmark footnote.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":84.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"terminal-bench-2-1","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gUXdlbiBDbGF1ZGUgQ29kZSBhdmdAMTAsNWggdGltZW91dCwxMzEwNzIgb3V0cHV0OyBjb21wYXJhdG9ycyBiZXN0IHB1Ymxpc2hlZCBhY3Jvc3MgaGFybmVzc2VzLg","id":"launch-qwen-552","ingestRunId":"launch-qwen","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"coding","comparabilityReason":"The Qwen card cites an external leaderboard or release report for this comparator; it is not a new Qwen evaluation. The source states that Fable 5 results may involve fallback execution; an exact configuration is not established.","comparable":false,"configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Qwen Claude Code avg@10,5h timeout,131072 output; comparators best published across harnesses.","family":"Terminal-Bench","locator":"Performance table: Terminal Bench 2.1","metric":"%","notes":"Cited comparator result; see the benchmark footnote.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":84.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"terminal-bench-2-1","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gUXdlbiBDbGF1ZGUgQ29kZSBhdmdAMTAsNWggdGltZW91dCwxMzEwNzIgb3V0cHV0OyBjb21wYXJhdG9ycyBiZXN0IHB1Ymxpc2hlZCBhY3Jvc3MgaGFybmVzc2VzLg","id":"launch-qwen-553","ingestRunId":"launch-qwen","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"coding","comparabilityReason":"The Qwen card cites an external leaderboard or release report for this comparator; it is not a new Qwen evaluation.","comparable":false,"configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Qwen Claude Code avg@10,5h timeout,131072 output; comparators best published across harnesses.","family":"Terminal-Bench","locator":"Performance table: Terminal Bench 2.1","metric":"%","notes":"Cited comparator result; see the benchmark footnote.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":88.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"terminal-bench-2-1","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gUXdlbiBDbGF1ZGUgQ29kZSBhdmdAMTAsNWggdGltZW91dCwxMzEwNzIgb3V0cHV0OyBjb21wYXJhdG9ycyBiZXN0IHB1Ymxpc2hlZCBhY3Jvc3MgaGFybmVzc2VzLg","id":"launch-qwen-554","ingestRunId":"launch-qwen","modelId":"qwen3.7-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"coding","comparabilityReason":"The Qwen card cites an external leaderboard or release report for this comparator; it is not a new Qwen evaluation.","comparable":false,"configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Qwen Claude Code avg@10,5h timeout,131072 output; comparators best published across harnesses.","family":"Terminal-Bench","locator":"Performance table: Terminal Bench 2.1","metric":"%","notes":"Cited comparator result; see the benchmark footnote.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":74.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"terminal-bench-2-1","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gUXdlbiBDbGF1ZGUgQ29kZSBhdmdAMTAsNWggdGltZW91dCwxMzEwNzIgb3V0cHV0OyBjb21wYXJhdG9ycyBiZXN0IHB1Ymxpc2hlZCBhY3Jvc3MgaGFybmVzc2VzLg","id":"launch-qwen-555","ingestRunId":"launch-qwen","modelId":"qwen-3.8-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"coding","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Qwen Claude Code avg@10,5h timeout,131072 output; comparators best published across harnesses.","family":"Terminal-Bench","locator":"Performance table: Terminal Bench 2.1","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":86.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"swe-bench-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"swe-bench-pro-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gUmVmaW5lZCB0YXNrIHNldCB3aXRoIHByb2JsZW1hdGljIHRhc2tzIGNvcnJlY3RlZDsgYWxsIGJhc2VsaW5lcyByZXJ1biwgQ2xhdWRlIENvZGUsdGVtcDEsdG9wX3AuOTUsMjU2SyBjb250ZXh0Lg","id":"launch-qwen-556","ingestRunId":"launch-qwen","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"coding","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Refined task set with problematic tasks corrected; all baselines rerun, Claude Code,temp1,top_p.95,256K context.","family":"SWE-bench Pro","locator":"Performance table: SWE-bench Pro","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":69.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"swe-bench-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"swe-bench-pro-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gUmVmaW5lZCB0YXNrIHNldCB3aXRoIHByb2JsZW1hdGljIHRhc2tzIGNvcnJlY3RlZDsgYWxsIGJhc2VsaW5lcyByZXJ1biwgQ2xhdWRlIENvZGUsdGVtcDEsdG9wX3AuOTUsMjU2SyBjb250ZXh0Lg","id":"launch-qwen-557","ingestRunId":"launch-qwen","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"coding","comparabilityReason":"The source states that Fable 5 results may involve fallback execution; an exact configuration is not established.","comparable":false,"configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Refined task set with problematic tasks corrected; all baselines rerun, Claude Code,temp1,top_p.95,256K context.","family":"SWE-bench Pro","locator":"Performance table: SWE-bench Pro","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":80,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"swe-bench-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"swe-bench-pro-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gUmVmaW5lZCB0YXNrIHNldCB3aXRoIHByb2JsZW1hdGljIHRhc2tzIGNvcnJlY3RlZDsgYWxsIGJhc2VsaW5lcyByZXJ1biwgQ2xhdWRlIENvZGUsdGVtcDEsdG9wX3AuOTUsMjU2SyBjb250ZXh0Lg","id":"launch-qwen-558","ingestRunId":"launch-qwen","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"coding","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Refined task set with problematic tasks corrected; all baselines rerun, Claude Code,temp1,top_p.95,256K context.","family":"SWE-bench Pro","locator":"Performance table: SWE-bench Pro","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":64.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"swe-bench-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"swe-bench-pro-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gUmVmaW5lZCB0YXNrIHNldCB3aXRoIHByb2JsZW1hdGljIHRhc2tzIGNvcnJlY3RlZDsgYWxsIGJhc2VsaW5lcyByZXJ1biwgQ2xhdWRlIENvZGUsdGVtcDEsdG9wX3AuOTUsMjU2SyBjb250ZXh0Lg","id":"launch-qwen-559","ingestRunId":"launch-qwen","modelId":"qwen3.7-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"coding","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Refined task set with problematic tasks corrected; all baselines rerun, Claude Code,temp1,top_p.95,256K context.","family":"SWE-bench Pro","locator":"Performance table: SWE-bench Pro","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":60.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"swe-bench-pro-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"swe-bench-pro-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gUmVmaW5lZCB0YXNrIHNldCB3aXRoIHByb2JsZW1hdGljIHRhc2tzIGNvcnJlY3RlZDsgYWxsIGJhc2VsaW5lcyByZXJ1biwgQ2xhdWRlIENvZGUsdGVtcDEsdG9wX3AuOTUsMjU2SyBjb250ZXh0Lg","id":"launch-qwen-560","ingestRunId":"launch-qwen","modelId":"qwen-3.8-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"coding","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Refined task set with problematic tasks corrected; all baselines rerun, Claude Code,temp1,top_p.95,256K context.","family":"SWE-bench Pro","locator":"Performance table: SWE-bench Pro","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":67.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"deepswe-1-1-pct","evidenceKind":"lab_self_report","harnessId":"deepswe-1-1-pct:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gQmVzdCBvZiBDbGF1ZGUgQ29kZSBhbmQgbWluaS1TV0UtYWdlbnQ7IFF3ZW4gYmVzdCBDbGF1ZGUgQ29kZTsgdGVtcDEsdG9wX3AuOTUsMjU2Sy4","id":"launch-qwen-561","ingestRunId":"launch-qwen","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"coding","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Best of Claude Code and mini-SWE-agent; Qwen best Claude Code; temp1,top_p.95,256K.","family":"DeepSWE 1.1","locator":"Performance table: DeepSWE 1.1","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":59,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"deepswe-1-1-pct","evidenceKind":"lab_self_report","harnessId":"deepswe-1-1-pct:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gQmVzdCBvZiBDbGF1ZGUgQ29kZSBhbmQgbWluaS1TV0UtYWdlbnQ7IFF3ZW4gYmVzdCBDbGF1ZGUgQ29kZTsgdGVtcDEsdG9wX3AuOTUsMjU2Sy4","id":"launch-qwen-562","ingestRunId":"launch-qwen","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"coding","comparabilityReason":"The source states that Fable 5 results may involve fallback execution; an exact configuration is not established.","comparable":false,"configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Best of Claude Code and mini-SWE-agent; Qwen best Claude Code; temp1,top_p.95,256K.","family":"DeepSWE 1.1","locator":"Performance table: DeepSWE 1.1","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":70,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"deepswe-1-1-pct","evidenceKind":"lab_self_report","harnessId":"deepswe-1-1-pct:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gQmVzdCBvZiBDbGF1ZGUgQ29kZSBhbmQgbWluaS1TV0UtYWdlbnQ7IFF3ZW4gYmVzdCBDbGF1ZGUgQ29kZTsgdGVtcDEsdG9wX3AuOTUsMjU2Sy4","id":"launch-qwen-563","ingestRunId":"launch-qwen","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"coding","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Best of Claude Code and mini-SWE-agent; Qwen best Claude Code; temp1,top_p.95,256K.","family":"DeepSWE 1.1","locator":"Performance table: DeepSWE 1.1","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":73,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"deepswe-1-1-pct","evidenceKind":"lab_self_report","harnessId":"deepswe-1-1-pct:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gQmVzdCBvZiBDbGF1ZGUgQ29kZSBhbmQgbWluaS1TV0UtYWdlbnQ7IFF3ZW4gYmVzdCBDbGF1ZGUgQ29kZTsgdGVtcDEsdG9wX3AuOTUsMjU2Sy4","id":"launch-qwen-564","ingestRunId":"launch-qwen","modelId":"qwen3.7-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"coding","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Best of Claude Code and mini-SWE-agent; Qwen best Claude Code; temp1,top_p.95,256K.","family":"DeepSWE 1.1","locator":"Performance table: DeepSWE 1.1","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":21.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"deepswe-1-1-pct","evidenceKind":"lab_self_report","harnessId":"deepswe-1-1-pct:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gQmVzdCBvZiBDbGF1ZGUgQ29kZSBhbmQgbWluaS1TV0UtYWdlbnQ7IFF3ZW4gYmVzdCBDbGF1ZGUgQ29kZTsgdGVtcDEsdG9wX3AuOTUsMjU2Sy4","id":"launch-qwen-565","ingestRunId":"launch-qwen","modelId":"qwen-3.8-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"coding","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Best of Claude Code and mini-SWE-agent; Qwen best Claude Code; temp1,top_p.95,256K.","family":"DeepSWE 1.1","locator":"Performance table: DeepSWE 1.1","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":56.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"nl2repo-bench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"nl2repo-bench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-566","ingestRunId":"launch-qwen","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"coding","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"NL2Repo-Bench","locator":"Performance table: NL2Repo-Bench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":69.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"nl2repo-bench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"nl2repo-bench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-567","ingestRunId":"launch-qwen","modelId":"qwen3.7-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"coding","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"NL2Repo-Bench","locator":"Performance table: NL2Repo-Bench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":47.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"nl2repo-bench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"nl2repo-bench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-568","ingestRunId":"launch-qwen","modelId":"qwen-3.8-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"coding","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"NL2Repo-Bench","locator":"Performance table: NL2Repo-Bench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":55.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"frontierswe-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"frontierswe-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-569","ingestRunId":"launch-qwen","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"coding","comparabilityReason":"The Qwen card cites an external leaderboard or release report for this comparator; it is not a new Qwen evaluation.","comparable":false,"configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"FrontierSWE","locator":"Performance table: FrontierSWE","metric":"%","notes":"Cited comparator result; see the benchmark footnote.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":70,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"frontierswe-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"frontierswe-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-570","ingestRunId":"launch-qwen","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"coding","comparabilityReason":"The Qwen card cites an external leaderboard or release report for this comparator; it is not a new Qwen evaluation. The source states that Fable 5 results may involve fallback execution; an exact configuration is not established.","comparable":false,"configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"FrontierSWE","locator":"Performance table: FrontierSWE","metric":"%","notes":"Cited comparator result; see the benchmark footnote.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":88.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"frontierswe-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"frontierswe-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-571","ingestRunId":"launch-qwen","modelId":"qwen3.7-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"coding","comparabilityReason":"The Qwen card cites an external leaderboard or release report for this comparator; it is not a new Qwen evaluation.","comparable":false,"configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"FrontierSWE","locator":"Performance table: FrontierSWE","metric":"%","notes":"Cited comparator result; see the benchmark footnote.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":40.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"frontierswe-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"frontierswe-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-572","ingestRunId":"launch-qwen","modelId":"qwen-3.8-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"coding","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"FrontierSWE","locator":"Performance table: FrontierSWE","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":73.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"mls-bench-lite-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mls-bench-lite-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-573","ingestRunId":"launch-qwen","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"coding","comparabilityReason":"The Qwen card cites an external leaderboard or release report for this comparator; it is not a new Qwen evaluation.","comparable":false,"configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"MLS-Bench-Lite","locator":"Performance table: MLS-Bench-Lite","metric":"%","notes":"Cited comparator result; see the benchmark footnote.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":42.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"mls-bench-lite-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mls-bench-lite-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-574","ingestRunId":"launch-qwen","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"coding","comparabilityReason":"The Qwen card cites an external leaderboard or release report for this comparator; it is not a new Qwen evaluation. The source states that Fable 5 results may involve fallback execution; an exact configuration is not established.","comparable":false,"configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"MLS-Bench-Lite","locator":"Performance table: MLS-Bench-Lite","metric":"%","notes":"Cited comparator result; see the benchmark footnote.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":49.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"mls-bench-lite-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mls-bench-lite-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-575","ingestRunId":"launch-qwen","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"coding","comparabilityReason":"The Qwen card cites an external leaderboard or release report for this comparator; it is not a new Qwen evaluation.","comparable":false,"configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"MLS-Bench-Lite","locator":"Performance table: MLS-Bench-Lite","metric":"%","notes":"Cited comparator result; see the benchmark footnote.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":46.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"mls-bench-lite-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mls-bench-lite-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-576","ingestRunId":"launch-qwen","modelId":"qwen3.7-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"coding","comparabilityReason":"The Qwen card cites an external leaderboard or release report for this comparator; it is not a new Qwen evaluation.","comparable":false,"configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"MLS-Bench-Lite","locator":"Performance table: MLS-Bench-Lite","metric":"%","notes":"Cited comparator result; see the benchmark footnote.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":31.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"mls-bench-lite-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mls-bench-lite-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-577","ingestRunId":"launch-qwen","modelId":"qwen-3.8-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"coding","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"MLS-Bench-Lite","locator":"Performance table: MLS-Bench-Lite","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":41,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"paperbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"paperbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gQmFzaWNBZ2VudCBDb2RlLURldjsgT3B1czQuNiBqdWRnZSwzcnVucywxMmggZWFjaC4","id":"launch-qwen-578","ingestRunId":"launch-qwen","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"coding","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. BasicAgent Code-Dev; Opus4.6 judge,3runs,12h each.","family":"PaperBench","locator":"Performance table: PaperBench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":80.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"paperbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"paperbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gQmFzaWNBZ2VudCBDb2RlLURldjsgT3B1czQuNiBqdWRnZSwzcnVucywxMmggZWFjaC4","id":"launch-qwen-579","ingestRunId":"launch-qwen","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"coding","comparabilityReason":"The source states that Fable 5 results may involve fallback execution; an exact configuration is not established.","comparable":false,"configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. BasicAgent Code-Dev; Opus4.6 judge,3runs,12h each.","family":"PaperBench","locator":"Performance table: PaperBench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":88.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"paperbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"paperbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gQmFzaWNBZ2VudCBDb2RlLURldjsgT3B1czQuNiBqdWRnZSwzcnVucywxMmggZWFjaC4","id":"launch-qwen-580","ingestRunId":"launch-qwen","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"coding","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. BasicAgent Code-Dev; Opus4.6 judge,3runs,12h each.","family":"PaperBench","locator":"Performance table: PaperBench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":90.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"paperbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"paperbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gQmFzaWNBZ2VudCBDb2RlLURldjsgT3B1czQuNiBqdWRnZSwzcnVucywxMmggZWFjaC4","id":"launch-qwen-581","ingestRunId":"launch-qwen","modelId":"qwen3.7-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"coding","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. BasicAgent Code-Dev; Opus4.6 judge,3runs,12h each.","family":"PaperBench","locator":"Performance table: PaperBench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":64.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"paperbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"paperbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gQmFzaWNBZ2VudCBDb2RlLURldjsgT3B1czQuNiBqdWRnZSwzcnVucywxMmggZWFjaC4","id":"launch-qwen-582","ingestRunId":"launch-qwen","modelId":"qwen-3.8-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"coding","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. BasicAgent Code-Dev; Opus4.6 judge,3runs,12h each.","family":"PaperBench","locator":"Performance table: PaperBench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":93,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"androidbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"androidbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-583","ingestRunId":"launch-qwen","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"coding","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"AndroidBench","locator":"Performance table: AndroidBench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":69.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"androidbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"androidbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-584","ingestRunId":"launch-qwen","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"coding","comparabilityReason":"The source states that Fable 5 results may involve fallback execution; an exact configuration is not established.","comparable":false,"configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"AndroidBench","locator":"Performance table: AndroidBench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":84.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"androidbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"androidbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-585","ingestRunId":"launch-qwen","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"coding","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"AndroidBench","locator":"Performance table: AndroidBench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":74,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"androidbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"androidbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-586","ingestRunId":"launch-qwen","modelId":"qwen3.7-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"coding","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"AndroidBench","locator":"Performance table: AndroidBench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":56.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"androidbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"androidbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-587","ingestRunId":"launch-qwen","modelId":"qwen-3.8-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"coding","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"AndroidBench","locator":"Performance table: AndroidBench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":75.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"qwenswebench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"qwenswebench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gSW50ZXJuYWwgc29mdHdhcmUgZW5naW5lZXJpbmcsQ2xhdWRlQ29kZSxhdmdAMyw4aCwzMjc2OCBvdXRwdXQsdGVtcDEsMjU2Sy4","id":"launch-qwen-588","ingestRunId":"launch-qwen","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"coding","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal software engineering,ClaudeCode,avg@3,8h,32768 output,temp1,256K.","family":"QwenSWEBench","locator":"Performance table: QwenSWEBench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":84,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"qwenswebench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"qwenswebench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gSW50ZXJuYWwgc29mdHdhcmUgZW5naW5lZXJpbmcsQ2xhdWRlQ29kZSxhdmdAMyw4aCwzMjc2OCBvdXRwdXQsdGVtcDEsMjU2Sy4","id":"launch-qwen-589","ingestRunId":"launch-qwen","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"coding","comparabilityReason":"The source states that Fable 5 results may involve fallback execution; an exact configuration is not established.","comparable":false,"configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal software engineering,ClaudeCode,avg@3,8h,32768 output,temp1,256K.","family":"QwenSWEBench","locator":"Performance table: QwenSWEBench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":86.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"qwenswebench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"qwenswebench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gSW50ZXJuYWwgc29mdHdhcmUgZW5naW5lZXJpbmcsQ2xhdWRlQ29kZSxhdmdAMyw4aCwzMjc2OCBvdXRwdXQsdGVtcDEsMjU2Sy4","id":"launch-qwen-590","ingestRunId":"launch-qwen","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"coding","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal software engineering,ClaudeCode,avg@3,8h,32768 output,temp1,256K.","family":"QwenSWEBench","locator":"Performance table: QwenSWEBench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":73.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"qwenswebench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"qwenswebench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gSW50ZXJuYWwgc29mdHdhcmUgZW5naW5lZXJpbmcsQ2xhdWRlQ29kZSxhdmdAMyw4aCwzMjc2OCBvdXRwdXQsdGVtcDEsMjU2Sy4","id":"launch-qwen-591","ingestRunId":"launch-qwen","modelId":"qwen3.7-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"coding","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal software engineering,ClaudeCode,avg@3,8h,32768 output,temp1,256K.","family":"QwenSWEBench","locator":"Performance table: QwenSWEBench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":63.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"qwenswebench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"qwenswebench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gSW50ZXJuYWwgc29mdHdhcmUgZW5naW5lZXJpbmcsQ2xhdWRlQ29kZSxhdmdAMyw4aCwzMjc2OCBvdXRwdXQsdGVtcDEsMjU2Sy4","id":"launch-qwen-592","ingestRunId":"launch-qwen","modelId":"qwen-3.8-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"coding","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal software engineering,ClaudeCode,avg@3,8h,32768 output,temp1,256K.","family":"QwenSWEBench","locator":"Performance table: QwenSWEBench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":80.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"qwenqoderbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"qwenqoderbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gSW50ZXJuYWwgUW9kZXIgdGFza3MsQ2xhdWRlQ29kZSxhdmdANSw2aCwzMjc2OCBvdXRwdXQsdGVtcDEsMjU2Sy4","id":"launch-qwen-593","ingestRunId":"launch-qwen","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal Qoder tasks,ClaudeCode,avg@5,6h,32768 output,temp1,256K.","family":"QwenQoderBench","locator":"Performance table: QwenQoderBench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":62.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"qwenqoderbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"qwenqoderbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gSW50ZXJuYWwgUW9kZXIgdGFza3MsQ2xhdWRlQ29kZSxhdmdANSw2aCwzMjc2OCBvdXRwdXQsdGVtcDEsMjU2Sy4","id":"launch-qwen-594","ingestRunId":"launch-qwen","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","comparabilityReason":"The source states that Fable 5 results may involve fallback execution; an exact configuration is not established.","comparable":false,"configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal Qoder tasks,ClaudeCode,avg@5,6h,32768 output,temp1,256K.","family":"QwenQoderBench","locator":"Performance table: QwenQoderBench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":63.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"qwenqoderbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"qwenqoderbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gSW50ZXJuYWwgUW9kZXIgdGFza3MsQ2xhdWRlQ29kZSxhdmdANSw2aCwzMjc2OCBvdXRwdXQsdGVtcDEsMjU2Sy4","id":"launch-qwen-595","ingestRunId":"launch-qwen","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal Qoder tasks,ClaudeCode,avg@5,6h,32768 output,temp1,256K.","family":"QwenQoderBench","locator":"Performance table: QwenQoderBench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":53.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"qwenqoderbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"qwenqoderbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gSW50ZXJuYWwgUW9kZXIgdGFza3MsQ2xhdWRlQ29kZSxhdmdANSw2aCwzMjc2OCBvdXRwdXQsdGVtcDEsMjU2Sy4","id":"launch-qwen-596","ingestRunId":"launch-qwen","modelId":"qwen3.7-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal Qoder tasks,ClaudeCode,avg@5,6h,32768 output,temp1,256K.","family":"QwenQoderBench","locator":"Performance table: QwenQoderBench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":36.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"qwenqoderbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"qwenqoderbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gSW50ZXJuYWwgUW9kZXIgdGFza3MsQ2xhdWRlQ29kZSxhdmdANSw2aCwzMjc2OCBvdXRwdXQsdGVtcDEsMjU2Sy4","id":"launch-qwen-597","ingestRunId":"launch-qwen","modelId":"qwen-3.8-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal Qoder tasks,ClaudeCode,avg@5,6h,32768 output,temp1,256K.","family":"QwenQoderBench","locator":"Performance table: QwenQoderBench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":58.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"qwenreactbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"qwenreactbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gSW50ZXJuYWwgYmlsaW5ndWFsIEVOL0NOIFJlYWN0IGJlbmNobWFyayw3Y2F0ZWdvcmllcyxDbGF1ZGVDb2RlLHJlbmRlcittdWx0aW1vZGFsanVkZ2UsQlQvRWxvLg","id":"launch-qwen-598","ingestRunId":"launch-qwen","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"coding","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal bilingual EN/CN React benchmark,7categories,ClaudeCode,render+multimodaljudge,BT/Elo.","family":"QwenReactBench","locator":"Performance table: QwenReactBench","metric":"Elo","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":1694,"scoreUnit":"elo","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"qwenreactbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"qwenreactbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gSW50ZXJuYWwgYmlsaW5ndWFsIEVOL0NOIFJlYWN0IGJlbmNobWFyayw3Y2F0ZWdvcmllcyxDbGF1ZGVDb2RlLHJlbmRlcittdWx0aW1vZGFsanVkZ2UsQlQvRWxvLg","id":"launch-qwen-599","ingestRunId":"launch-qwen","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"coding","comparabilityReason":"The source states that Fable 5 results may involve fallback execution; an exact configuration is not established.","comparable":false,"configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal bilingual EN/CN React benchmark,7categories,ClaudeCode,render+multimodaljudge,BT/Elo.","family":"QwenReactBench","locator":"Performance table: QwenReactBench","metric":"Elo","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":1770,"scoreUnit":"elo","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"qwenreactbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"qwenreactbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gSW50ZXJuYWwgYmlsaW5ndWFsIEVOL0NOIFJlYWN0IGJlbmNobWFyayw3Y2F0ZWdvcmllcyxDbGF1ZGVDb2RlLHJlbmRlcittdWx0aW1vZGFsanVkZ2UsQlQvRWxvLg","id":"launch-qwen-600","ingestRunId":"launch-qwen","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"coding","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal bilingual EN/CN React benchmark,7categories,ClaudeCode,render+multimodaljudge,BT/Elo.","family":"QwenReactBench","locator":"Performance table: QwenReactBench","metric":"Elo","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":1564,"scoreUnit":"elo","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"qwenreactbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"qwenreactbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gSW50ZXJuYWwgYmlsaW5ndWFsIEVOL0NOIFJlYWN0IGJlbmNobWFyayw3Y2F0ZWdvcmllcyxDbGF1ZGVDb2RlLHJlbmRlcittdWx0aW1vZGFsanVkZ2UsQlQvRWxvLg","id":"launch-qwen-601","ingestRunId":"launch-qwen","modelId":"qwen3.7-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"coding","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal bilingual EN/CN React benchmark,7categories,ClaudeCode,render+multimodaljudge,BT/Elo.","family":"QwenReactBench","locator":"Performance table: QwenReactBench","metric":"Elo","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":1538,"scoreUnit":"elo","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"qwenreactbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"qwenreactbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gSW50ZXJuYWwgYmlsaW5ndWFsIEVOL0NOIFJlYWN0IGJlbmNobWFyayw3Y2F0ZWdvcmllcyxDbGF1ZGVDb2RlLHJlbmRlcittdWx0aW1vZGFsanVkZ2UsQlQvRWxvLg","id":"launch-qwen-602","ingestRunId":"launch-qwen","modelId":"qwen-3.8-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"coding","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal bilingual EN/CN React benchmark,7categories,ClaudeCode,render+multimodaljudge,BT/Elo.","family":"QwenReactBench","locator":"Performance table: QwenReactBench","metric":"Elo","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":1724,"scoreUnit":"elo","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"qwensvgbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"qwensvgbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gSW50ZXJuYWwgYmlsaW5ndWFsIEVOL0NOIFNWRyBiZW5jaG1hcmsscmVuZGVyK211bHRpbW9kYWxqdWRnZSxCVC9FbG8u","id":"launch-qwen-603","ingestRunId":"launch-qwen","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"coding","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal bilingual EN/CN SVG benchmark,render+multimodaljudge,BT/Elo.","family":"QwenSVGBench","locator":"Performance table: QwenSVGBench","metric":"Elo","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":1648,"scoreUnit":"elo","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"qwensvgbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"qwensvgbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gSW50ZXJuYWwgYmlsaW5ndWFsIEVOL0NOIFNWRyBiZW5jaG1hcmsscmVuZGVyK211bHRpbW9kYWxqdWRnZSxCVC9FbG8u","id":"launch-qwen-604","ingestRunId":"launch-qwen","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"coding","comparabilityReason":"The source states that Fable 5 results may involve fallback execution; an exact configuration is not established.","comparable":false,"configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal bilingual EN/CN SVG benchmark,render+multimodaljudge,BT/Elo.","family":"QwenSVGBench","locator":"Performance table: QwenSVGBench","metric":"Elo","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":1690,"scoreUnit":"elo","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"qwensvgbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"qwensvgbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gSW50ZXJuYWwgYmlsaW5ndWFsIEVOL0NOIFNWRyBiZW5jaG1hcmsscmVuZGVyK211bHRpbW9kYWxqdWRnZSxCVC9FbG8u","id":"launch-qwen-605","ingestRunId":"launch-qwen","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"coding","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal bilingual EN/CN SVG benchmark,render+multimodaljudge,BT/Elo.","family":"QwenSVGBench","locator":"Performance table: QwenSVGBench","metric":"Elo","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":1758,"scoreUnit":"elo","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"qwensvgbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"qwensvgbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gSW50ZXJuYWwgYmlsaW5ndWFsIEVOL0NOIFNWRyBiZW5jaG1hcmsscmVuZGVyK211bHRpbW9kYWxqdWRnZSxCVC9FbG8u","id":"launch-qwen-606","ingestRunId":"launch-qwen","modelId":"qwen3.7-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"coding","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal bilingual EN/CN SVG benchmark,render+multimodaljudge,BT/Elo.","family":"QwenSVGBench","locator":"Performance table: QwenSVGBench","metric":"Elo","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":1499,"scoreUnit":"elo","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"qwensvgbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"qwensvgbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gSW50ZXJuYWwgYmlsaW5ndWFsIEVOL0NOIFNWRyBiZW5jaG1hcmsscmVuZGVyK211bHRpbW9kYWxqdWRnZSxCVC9FbG8u","id":"launch-qwen-607","ingestRunId":"launch-qwen","modelId":"qwen-3.8-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"coding","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal bilingual EN/CN SVG benchmark,render+multimodaljudge,BT/Elo.","family":"QwenSVGBench","locator":"Performance table: QwenSVGBench","metric":"Elo","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":1713,"scoreUnit":"elo","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"coworkbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"coworkbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gSW50ZXJuYWwgcHJvZmVzc2lvbmFsIHdvcmsgdGFza3MgYWNyb3NzIHNjaWVuY2UsZmluYW5jZSxsYXcsbWVkaWNhbCxwcm9kdWN0aXZpdHku","id":"launch-qwen-608","ingestRunId":"launch-qwen","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal professional work tasks across science,finance,law,medical,productivity.","family":"CoWorkBench","locator":"Performance table: CoWorkBench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":72.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"coworkbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"coworkbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gSW50ZXJuYWwgcHJvZmVzc2lvbmFsIHdvcmsgdGFza3MgYWNyb3NzIHNjaWVuY2UsZmluYW5jZSxsYXcsbWVkaWNhbCxwcm9kdWN0aXZpdHku","id":"launch-qwen-609","ingestRunId":"launch-qwen","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","comparabilityReason":"The source states that Fable 5 results may involve fallback execution; an exact configuration is not established.","comparable":false,"configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal professional work tasks across science,finance,law,medical,productivity.","family":"CoWorkBench","locator":"Performance table: CoWorkBench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":75.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"coworkbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"coworkbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gSW50ZXJuYWwgcHJvZmVzc2lvbmFsIHdvcmsgdGFza3MgYWNyb3NzIHNjaWVuY2UsZmluYW5jZSxsYXcsbWVkaWNhbCxwcm9kdWN0aXZpdHku","id":"launch-qwen-610","ingestRunId":"launch-qwen","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal professional work tasks across science,finance,law,medical,productivity.","family":"CoWorkBench","locator":"Performance table: CoWorkBench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":71.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"coworkbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"coworkbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gSW50ZXJuYWwgcHJvZmVzc2lvbmFsIHdvcmsgdGFza3MgYWNyb3NzIHNjaWVuY2UsZmluYW5jZSxsYXcsbWVkaWNhbCxwcm9kdWN0aXZpdHku","id":"launch-qwen-611","ingestRunId":"launch-qwen","modelId":"qwen3.7-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal professional work tasks across science,finance,law,medical,productivity.","family":"CoWorkBench","locator":"Performance table: CoWorkBench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":64.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"coworkbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"coworkbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gSW50ZXJuYWwgcHJvZmVzc2lvbmFsIHdvcmsgdGFza3MgYWNyb3NzIHNjaWVuY2UsZmluYW5jZSxsYXcsbWVkaWNhbCxwcm9kdWN0aXZpdHku","id":"launch-qwen-612","ingestRunId":"launch-qwen","modelId":"qwen-3.8-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal professional work tasks across science,finance,law,medical,productivity.","family":"CoWorkBench","locator":"Performance table: CoWorkBench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":74.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"workspacebench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"workspacebench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-613","ingestRunId":"launch-qwen","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"WorkSpaceBench","locator":"Performance table: WorkSpaceBench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":66.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"workspacebench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"workspacebench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-614","ingestRunId":"launch-qwen","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","comparabilityReason":"The source states that Fable 5 results may involve fallback execution; an exact configuration is not established.","comparable":false,"configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"WorkSpaceBench","locator":"Performance table: WorkSpaceBench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":68.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"workspacebench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"workspacebench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-615","ingestRunId":"launch-qwen","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"WorkSpaceBench","locator":"Performance table: WorkSpaceBench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":65.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"workspacebench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"workspacebench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-616","ingestRunId":"launch-qwen","modelId":"qwen3.7-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"WorkSpaceBench","locator":"Performance table: WorkSpaceBench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":61.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"workspacebench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"workspacebench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-617","ingestRunId":"launch-qwen","modelId":"qwen-3.8-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"WorkSpaceBench","locator":"Performance table: WorkSpaceBench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":67.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"jobbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"jobbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-618","ingestRunId":"launch-qwen","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"JobBench","locator":"Performance table: JobBench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":48.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"jobbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"jobbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-619","ingestRunId":"launch-qwen","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","comparabilityReason":"The source states that Fable 5 results may involve fallback execution; an exact configuration is not established.","comparable":false,"configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"JobBench","locator":"Performance table: JobBench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":57.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"jobbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"jobbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-620","ingestRunId":"launch-qwen","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"JobBench","locator":"Performance table: JobBench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":45.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"jobbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"jobbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-621","ingestRunId":"launch-qwen","modelId":"qwen3.7-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"JobBench","locator":"Performance table: JobBench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":31.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"jobbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"jobbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-622","ingestRunId":"launch-qwen","modelId":"qwen-3.8-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"JobBench","locator":"Performance table: JobBench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":53.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"skillsbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"skillsbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gdjEuMSBwdWJsaWM4N3Rhc2tzLDNydW5zOyBBbnRocm9waWMgQ2xhdWRlQ29kZSxPcGVuQUkgQ29kZXgsUXdlbiBPcGVuQ29kZS4","id":"launch-qwen-623","ingestRunId":"launch-qwen","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. v1.1 public87tasks,3runs; Anthropic ClaudeCode,OpenAI Codex,Qwen OpenCode.","family":"SkillsBench","locator":"Performance table: SkillsBench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":65.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"skillsbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"skillsbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gdjEuMSBwdWJsaWM4N3Rhc2tzLDNydW5zOyBBbnRocm9waWMgQ2xhdWRlQ29kZSxPcGVuQUkgQ29kZXgsUXdlbiBPcGVuQ29kZS4","id":"launch-qwen-624","ingestRunId":"launch-qwen","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","comparabilityReason":"The source states that Fable 5 results may involve fallback execution; an exact configuration is not established.","comparable":false,"configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. v1.1 public87tasks,3runs; Anthropic ClaudeCode,OpenAI Codex,Qwen OpenCode.","family":"SkillsBench","locator":"Performance table: SkillsBench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":70.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"skillsbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"skillsbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gdjEuMSBwdWJsaWM4N3Rhc2tzLDNydW5zOyBBbnRocm9waWMgQ2xhdWRlQ29kZSxPcGVuQUkgQ29kZXgsUXdlbiBPcGVuQ29kZS4","id":"launch-qwen-625","ingestRunId":"launch-qwen","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. v1.1 public87tasks,3runs; Anthropic ClaudeCode,OpenAI Codex,Qwen OpenCode.","family":"SkillsBench","locator":"Performance table: SkillsBench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":73.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"skillsbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"skillsbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gdjEuMSBwdWJsaWM4N3Rhc2tzLDNydW5zOyBBbnRocm9waWMgQ2xhdWRlQ29kZSxPcGVuQUkgQ29kZXgsUXdlbiBPcGVuQ29kZS4","id":"launch-qwen-626","ingestRunId":"launch-qwen","modelId":"qwen3.7-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. v1.1 public87tasks,3runs; Anthropic ClaudeCode,OpenAI Codex,Qwen OpenCode.","family":"SkillsBench","locator":"Performance table: SkillsBench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":61.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"skillsbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"skillsbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gdjEuMSBwdWJsaWM4N3Rhc2tzLDNydW5zOyBBbnRocm9waWMgQ2xhdWRlQ29kZSxPcGVuQUkgQ29kZXgsUXdlbiBPcGVuQ29kZS4","id":"launch-qwen-627","ingestRunId":"launch-qwen","modelId":"qwen-3.8-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. v1.1 public87tasks,3runs; Anthropic ClaudeCode,OpenAI Codex,Qwen OpenCode.","family":"SkillsBench","locator":"Performance table: SkillsBench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":70.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"agents-last-exam-pass-score-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"agents-last-exam-pass-score-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gUGFzcw","id":"launch-qwen-628","ingestRunId":"launch-qwen","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Pass","family":"Agents' Last Exam (Pass / Score)","locator":"Performance table: Agents' Last Exam (Pass / Score)","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":27,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"agents-last-exam-pass-score-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"agents-last-exam-pass-score-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gU2NvcmU","id":"launch-qwen-629","ingestRunId":"launch-qwen","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Score","family":"Agents' Last Exam (Pass / Score)","locator":"Performance table: Agents' Last Exam (Pass / Score)","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":45.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"agents-last-exam-pass-score-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"agents-last-exam-pass-score-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gUGFzcw","id":"launch-qwen-630","ingestRunId":"launch-qwen","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Pass","family":"Agents' Last Exam (Pass / Score)","locator":"Performance table: Agents' Last Exam (Pass / Score)","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":30.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"agents-last-exam-pass-score-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"agents-last-exam-pass-score-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gU2NvcmU","id":"launch-qwen-631","ingestRunId":"launch-qwen","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Score","family":"Agents' Last Exam (Pass / Score)","locator":"Performance table: Agents' Last Exam (Pass / Score)","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":53.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"agents-last-exam-pass-score-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"agents-last-exam-pass-score-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gUGFzcw","id":"launch-qwen-632","ingestRunId":"launch-qwen","modelId":"qwen3.7-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Pass","family":"Agents' Last Exam (Pass / Score)","locator":"Performance table: Agents' Last Exam (Pass / Score)","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":11.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"agents-last-exam-pass-score-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"agents-last-exam-pass-score-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gU2NvcmU","id":"launch-qwen-633","ingestRunId":"launch-qwen","modelId":"qwen3.7-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Score","family":"Agents' Last Exam (Pass / Score)","locator":"Performance table: Agents' Last Exam (Pass / Score)","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":31.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"agents-last-exam-pass-score-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"agents-last-exam-pass-score-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gUGFzcw","id":"launch-qwen-634","ingestRunId":"launch-qwen","modelId":"qwen-3.8-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Pass","family":"Agents' Last Exam (Pass / Score)","locator":"Performance table: Agents' Last Exam (Pass / Score)","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":27,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"agents-last-exam-pass-score-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"agents-last-exam-pass-score-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gU2NvcmU","id":"launch-qwen-635","ingestRunId":"launch-qwen","modelId":"qwen-3.8-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Score","family":"Agents' Last Exam (Pass / Score)","locator":"Performance table: Agents' Last Exam (Pass / Score)","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":52.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"automation-bench-pass-1-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"automation-bench-pass-1-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-636","ingestRunId":"launch-qwen","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"Automation-Bench (Pass@1)","locator":"Performance table: Automation-Bench (Pass@1)","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":27.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"automation-bench-pass-1-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"automation-bench-pass-1-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-637","ingestRunId":"launch-qwen","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","comparabilityReason":"The source states that Fable 5 results may involve fallback execution; an exact configuration is not established.","comparable":false,"configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"Automation-Bench (Pass@1)","locator":"Performance table: Automation-Bench (Pass@1)","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":29.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"automation-bench-pass-1-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"automation-bench-pass-1-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-638","ingestRunId":"launch-qwen","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"Automation-Bench (Pass@1)","locator":"Performance table: Automation-Bench (Pass@1)","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":29.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"automation-bench-pass-1-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"automation-bench-pass-1-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-639","ingestRunId":"launch-qwen","modelId":"qwen3.7-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"Automation-Bench (Pass@1)","locator":"Performance table: Automation-Bench (Pass@1)","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":14.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"automation-bench-pass-1-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"automation-bench-pass-1-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-640","ingestRunId":"launch-qwen","modelId":"qwen-3.8-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"Automation-Bench (Pass@1)","locator":"Performance table: Automation-Bench (Pass@1)","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":27.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"toolathlon-verified-pass-1-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"toolathlon-verified-pass-1-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-641","ingestRunId":"launch-qwen","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"Toolathlon Verified (Pass@1)","locator":"Performance table: Toolathlon Verified (Pass@1)","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":76.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"toolathlon-verified-pass-1-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"toolathlon-verified-pass-1-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-642","ingestRunId":"launch-qwen","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","comparabilityReason":"The source states that Fable 5 results may involve fallback execution; an exact configuration is not established.","comparable":false,"configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"Toolathlon Verified (Pass@1)","locator":"Performance table: Toolathlon Verified (Pass@1)","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":77.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"toolathlon-verified-pass-1-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"toolathlon-verified-pass-1-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-643","ingestRunId":"launch-qwen","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"Toolathlon Verified (Pass@1)","locator":"Performance table: Toolathlon Verified (Pass@1)","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":74.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"toolathlon-verified-pass-1-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"toolathlon-verified-pass-1-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-644","ingestRunId":"launch-qwen","modelId":"qwen3.7-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"Toolathlon Verified (Pass@1)","locator":"Performance table: Toolathlon Verified (Pass@1)","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":49.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"toolathlon-verified-pass-1-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"toolathlon-verified-pass-1-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-645","ingestRunId":"launch-qwen","modelId":"qwen-3.8-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"Toolathlon Verified (Pass@1)","locator":"Performance table: Toolathlon Verified (Pass@1)","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":72.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"widesearch-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"widesearch-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gSXRlbS1GMSBvdmVyNHJ1bnM7IFF3ZW4tQWdlbnQgZm9yIFF3ZW4sQ2xhdWRlQ29kZSBmb3IgY29tcGFyYXRvcnMu","id":"launch-qwen-646","ingestRunId":"launch-qwen","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Item-F1 over4runs; Qwen-Agent for Qwen,ClaudeCode for comparators.","family":"WideSearch","locator":"Performance table: WideSearch","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":72.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"widesearch-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"widesearch-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gSXRlbS1GMSBvdmVyNHJ1bnM7IFF3ZW4tQWdlbnQgZm9yIFF3ZW4sQ2xhdWRlQ29kZSBmb3IgY29tcGFyYXRvcnMu","id":"launch-qwen-647","ingestRunId":"launch-qwen","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","comparabilityReason":"The source states that Fable 5 results may involve fallback execution; an exact configuration is not established.","comparable":false,"configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Item-F1 over4runs; Qwen-Agent for Qwen,ClaudeCode for comparators.","family":"WideSearch","locator":"Performance table: WideSearch","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":81.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"widesearch-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"widesearch-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gSXRlbS1GMSBvdmVyNHJ1bnM7IFF3ZW4tQWdlbnQgZm9yIFF3ZW4sQ2xhdWRlQ29kZSBmb3IgY29tcGFyYXRvcnMu","id":"launch-qwen-648","ingestRunId":"launch-qwen","modelId":"qwen3.7-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Item-F1 over4runs; Qwen-Agent for Qwen,ClaudeCode for comparators.","family":"WideSearch","locator":"Performance table: WideSearch","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":75.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"widesearch-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"widesearch-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4gSXRlbS1GMSBvdmVyNHJ1bnM7IFF3ZW4tQWdlbnQgZm9yIFF3ZW4sQ2xhdWRlQ29kZSBmb3IgY29tcGFyYXRvcnMu","id":"launch-qwen-649","ingestRunId":"launch-qwen","modelId":"qwen-3.8-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Item-F1 over4runs; Qwen-Agent for Qwen,ClaudeCode for comparators.","family":"WideSearch","locator":"Performance table: WideSearch","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":81.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"hle-w-tools-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"hle-w-tools-source-release-snapshot-version-not-specified-pct:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-650","ingestRunId":"launch-qwen","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"hard_reasoning","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"Humanity's Last Exam","locator":"Performance table: HLE w/ tools","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":57.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"hle-w-tools-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"hle-w-tools-source-release-snapshot-version-not-specified-pct:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-651","ingestRunId":"launch-qwen","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"hard_reasoning","comparabilityReason":"The source states that Fable 5 results may involve fallback execution; an exact configuration is not established.","comparable":false,"configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"Humanity's Last Exam","locator":"Performance table: HLE w/ tools","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":64.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"hle-w-tools-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"hle-w-tools-source-release-snapshot-version-not-specified-pct:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-652","ingestRunId":"launch-qwen","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"hard_reasoning","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"Humanity's Last Exam","locator":"Performance table: HLE w/ tools","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":58,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"hle-w-tools-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"hle-w-tools-source-release-snapshot-version-not-specified-pct:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-653","ingestRunId":"launch-qwen","modelId":"qwen3.7-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"hard_reasoning","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"Humanity's Last Exam","locator":"Performance table: HLE w/ tools","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":53.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"hle-w-tools-source-release-snapshot-version-not-specified-pct","evidenceKind":"lab_self_report","harnessId":"hle-w-tools-source-release-snapshot-version-not-specified-pct:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-654","ingestRunId":"launch-qwen","modelId":"qwen-3.8-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"hard_reasoning","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"Humanity's Last Exam","locator":"Performance table: HLE w/ tools","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":56.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"gpqa-diamond-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gpqa-diamond-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-655","ingestRunId":"launch-qwen","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"hard_reasoning","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"GPQA Diamond","locator":"Performance table: GPQA Diamond","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":92,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"gpqa-diamond-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gpqa-diamond-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-656","ingestRunId":"launch-qwen","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"hard_reasoning","comparabilityReason":"The source states that Fable 5 results may involve fallback execution; an exact configuration is not established.","comparable":false,"configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"GPQA Diamond","locator":"Performance table: GPQA Diamond","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":92.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"gpqa-diamond-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gpqa-diamond-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-657","ingestRunId":"launch-qwen","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"hard_reasoning","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"GPQA Diamond","locator":"Performance table: GPQA Diamond","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":94.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"gpqa-diamond-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gpqa-diamond-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-658","ingestRunId":"launch-qwen","modelId":"qwen3.7-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"hard_reasoning","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"GPQA Diamond","locator":"Performance table: GPQA Diamond","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":92.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"gpqa-diamond-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gpqa-diamond-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-659","ingestRunId":"launch-qwen","modelId":"qwen-3.8-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"hard_reasoning","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"GPQA Diamond","locator":"Performance table: GPQA Diamond","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":92.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"hle-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hle-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-660","ingestRunId":"launch-qwen","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"hard_reasoning","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"Humanity's Last Exam","locator":"Performance table: HLE","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":45.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"hle-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hle-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-661","ingestRunId":"launch-qwen","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"hard_reasoning","comparabilityReason":"The source states that Fable 5 results may involve fallback execution; an exact configuration is not established.","comparable":false,"configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"Humanity's Last Exam","locator":"Performance table: HLE","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":53.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"hle-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hle-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-662","ingestRunId":"launch-qwen","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"hard_reasoning","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"Humanity's Last Exam","locator":"Performance table: HLE","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":47.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"hle-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hle-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-663","ingestRunId":"launch-qwen","modelId":"qwen3.7-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"hard_reasoning","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"Humanity's Last Exam","locator":"Performance table: HLE","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":41.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"hle-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hle-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-664","ingestRunId":"launch-qwen","modelId":"qwen-3.8-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"hard_reasoning","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"Humanity's Last Exam","locator":"Performance table: HLE","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":43.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"ifbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"ifbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-665","ingestRunId":"launch-qwen","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"human_pref","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"IFBench","locator":"Performance table: IFBench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":62.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"ifbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"ifbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-666","ingestRunId":"launch-qwen","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"human_pref","comparabilityReason":"The source states that Fable 5 results may involve fallback execution; an exact configuration is not established.","comparable":false,"configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"IFBench","locator":"Performance table: IFBench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":63.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"ifbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"ifbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-667","ingestRunId":"launch-qwen","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"human_pref","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"IFBench","locator":"Performance table: IFBench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":72.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"ifbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"ifbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-668","ingestRunId":"launch-qwen","modelId":"qwen3.7-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"human_pref","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"IFBench","locator":"Performance table: IFBench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":79.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"ifbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"ifbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-669","ingestRunId":"launch-qwen","modelId":"qwen-3.8-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"human_pref","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"IFBench","locator":"Performance table: IFBench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":82.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"onemillion-bench-expert-score-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"onemillion-bench-expert-score-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-670","ingestRunId":"launch-qwen","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"$OneMillion-Bench (expert score)","locator":"Performance table: $OneMillion-Bench (expert score)","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":41.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"onemillion-bench-expert-score-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"onemillion-bench-expert-score-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-671","ingestRunId":"launch-qwen","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","comparabilityReason":"The source states that Fable 5 results may involve fallback execution; an exact configuration is not established.","comparable":false,"configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"$OneMillion-Bench (expert score)","locator":"Performance table: $OneMillion-Bench (expert score)","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":55.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"onemillion-bench-expert-score-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"onemillion-bench-expert-score-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-672","ingestRunId":"launch-qwen","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"$OneMillion-Bench (expert score)","locator":"Performance table: $OneMillion-Bench (expert score)","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":53.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"onemillion-bench-expert-score-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"onemillion-bench-expert-score-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-673","ingestRunId":"launch-qwen","modelId":"qwen3.7-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"$OneMillion-Bench (expert score)","locator":"Performance table: $OneMillion-Bench (expert score)","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":44.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"onemillion-bench-expert-score-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"onemillion-bench-expert-score-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-674","ingestRunId":"launch-qwen","modelId":"qwen-3.8-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"$OneMillion-Bench (expert score)","locator":"Performance table: $OneMillion-Bench (expert score)","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":52.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"healthbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"healthbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-675","ingestRunId":"launch-qwen","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"knowledge","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"HealthBench","locator":"Performance table: HealthBench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":52.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"healthbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"healthbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-676","ingestRunId":"launch-qwen","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"knowledge","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"HealthBench","locator":"Performance table: HealthBench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":55.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"healthbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"healthbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-677","ingestRunId":"launch-qwen","modelId":"qwen3.7-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"knowledge","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"HealthBench","locator":"Performance table: HealthBench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":54.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"healthbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"healthbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-678","ingestRunId":"launch-qwen","modelId":"qwen-3.8-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"knowledge","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"HealthBench","locator":"Performance table: HealthBench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":60.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"plawbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"plawbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-679","ingestRunId":"launch-qwen","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"PLawBench","locator":"Performance table: PLawBench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":69.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"plawbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"plawbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-680","ingestRunId":"launch-qwen","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","comparabilityReason":"The source states that Fable 5 results may involve fallback execution; an exact configuration is not established.","comparable":false,"configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"PLawBench","locator":"Performance table: PLawBench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":70.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"plawbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"plawbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-681","ingestRunId":"launch-qwen","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"PLawBench","locator":"Performance table: PLawBench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":72.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"plawbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"plawbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-682","ingestRunId":"launch-qwen","modelId":"qwen3.7-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"PLawBench","locator":"Performance table: PLawBench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":58.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"plawbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"plawbench-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-683","ingestRunId":"launch-qwen","modelId":"qwen-3.8-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"PLawBench","locator":"Performance table: PLawBench","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":73.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"prbench-legal-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"prbench-legal-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-684","ingestRunId":"launch-qwen","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"PRBench-Legal","locator":"Performance table: PRBench-Legal","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":52.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"prbench-legal-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"prbench-legal-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-685","ingestRunId":"launch-qwen","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","comparabilityReason":"The source states that Fable 5 results may involve fallback execution; an exact configuration is not established.","comparable":false,"configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"PRBench-Legal","locator":"Performance table: PRBench-Legal","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":57.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"prbench-legal-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"prbench-legal-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-686","ingestRunId":"launch-qwen","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"PRBench-Legal","locator":"Performance table: PRBench-Legal","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":57.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"prbench-legal-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"prbench-legal-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-687","ingestRunId":"launch-qwen","modelId":"qwen3.7-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"PRBench-Legal","locator":"Performance table: PRBench-Legal","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":48.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"prbench-legal-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"prbench-legal-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-688","ingestRunId":"launch-qwen","modelId":"qwen-3.8-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"PRBench-Legal","locator":"Performance table: PRBench-Legal","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":57.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"prbench-finance-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"prbench-finance-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-689","ingestRunId":"launch-qwen","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"PRBench-Finance","locator":"Performance table: PRBench-Finance","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":51.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"prbench-finance-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"prbench-finance-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-690","ingestRunId":"launch-qwen","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","comparabilityReason":"The source states that Fable 5 results may involve fallback execution; an exact configuration is not established.","comparable":false,"configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"PRBench-Finance","locator":"Performance table: PRBench-Finance","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":55.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"prbench-finance-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"prbench-finance-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-691","ingestRunId":"launch-qwen","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"PRBench-Finance","locator":"Performance table: PRBench-Finance","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":55.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"prbench-finance-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"prbench-finance-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-692","ingestRunId":"launch-qwen","modelId":"qwen3.7-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"PRBench-Finance","locator":"Performance table: PRBench-Finance","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":46.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"prbench-finance-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"prbench-finance-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-693","ingestRunId":"launch-qwen","modelId":"qwen-3.8-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"agentic","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"PRBench-Finance","locator":"Performance table: PRBench-Finance","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":58.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"mrcr-v2-256k-8-needle-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mrcr-v2-256k-8-needle-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-694","ingestRunId":"launch-qwen","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"long_context","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"MRCR v2 256K (8-needle)","locator":"Performance table: MRCR v2 256K (8-needle)","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":83.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"mrcr-v2-256k-8-needle-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mrcr-v2-256k-8-needle-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-695","ingestRunId":"launch-qwen","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"long_context","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"MRCR v2 256K (8-needle)","locator":"Performance table: MRCR v2 256K (8-needle)","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":93.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"mrcr-v2-256k-8-needle-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mrcr-v2-256k-8-needle-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-696","ingestRunId":"launch-qwen","modelId":"qwen3.7-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"long_context","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"MRCR v2 256K (8-needle)","locator":"Performance table: MRCR v2 256K (8-needle)","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":86.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"mrcr-v2-256k-8-needle-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"mrcr-v2-256k-8-needle-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-697","ingestRunId":"launch-qwen","modelId":"qwen-3.8-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"long_context","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"MRCR v2 256K (8-needle)","locator":"Performance table: MRCR v2 256K (8-needle)","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":92.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"longbench-v2-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"longbench-v2-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-698","ingestRunId":"launch-qwen","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"long_context","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"LongBench v2","locator":"Performance table: LongBench v2","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":69.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"longbench-v2-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"longbench-v2-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-699","ingestRunId":"launch-qwen","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"long_context","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"LongBench v2","locator":"Performance table: LongBench v2","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":67.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"longbench-v2-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"longbench-v2-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-700","ingestRunId":"launch-qwen","modelId":"qwen3.7-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"long_context","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"LongBench v2","locator":"Performance table: LongBench v2","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":65.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"longbench-v2-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"longbench-v2-source-release-snapshot-version-not-specified:qwen:U291cmNlIGJlbmNobWFyay1zcGVjaWZpYyBtZXRob2RvbG9neTsgUXdlbiBiZW5jaG1hcmsgZWZmb3J0IHVuc3BlY2lmaWVkLCBHUFQtNS42IFNvbCBtYXggcGVyIHRhYmxlIGhlYWRlci4gTWF4IGluIFF3ZW4gbW9kZWwgbmFtZXMgaXMgbm90IGEgcmVwb3J0ZWQgZWZmb3J0IHNldHRpbmcuIENvbXBhcmF0b3Igc2V0dGluZ3MgYW5kIHNvdXJjZSBjaXRhdGlvbnMgYXMgZG9jdW1lbnRlZCBpbiB0aGUgbW9kZWwgY2FyZC4","id":"launch-qwen-701","ingestRunId":"launch-qwen","modelId":"qwen-3.8-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"7690c680c2d6eb284179ef5dd4db63260d248531cae9087b09b2c2200230f331","report":{"category":"long_context","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card.","family":"LongBench v2","locator":"Performance table: LongBench v2","metric":"%","notes":"First-party reported result.","sourceId":"qwen","sourceTitle":"Qwen/Qwen3.8-2.4T-A95B"},"score":66.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md"},{"benchmarkId":"aa-intelligence-index-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"aa-intelligence-index-source-release-snapshot-version-not-specified:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","id":"launch-xai-1318","ingestRunId":"launch-xai","modelId":"grok-4.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI.","family":"AA Intelligence Index","locator":"Performance table","metric":"index points","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"xai","sourceTitle":"Grok 4.6 launch evaluations"},"score":61,"scoreUnit":"index","sourceUrl":"https://x.ai/news/grok-4-6"},{"benchmarkId":"aa-intelligence-index-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"aa-intelligence-index-source-release-snapshot-version-not-specified:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","id":"launch-xai-1319","ingestRunId":"launch-xai","modelId":"grok-4.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI.","family":"AA Intelligence Index","locator":"Performance table","metric":"index points","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"xai","sourceTitle":"Grok 4.6 launch evaluations"},"score":56,"scoreUnit":"index","sourceUrl":"https://x.ai/news/grok-4-6"},{"benchmarkId":"aa-intelligence-index-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"aa-intelligence-index-source-release-snapshot-version-not-specified:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","id":"launch-xai-1320","ingestRunId":"launch-xai","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI.","family":"AA Intelligence Index","locator":"Performance table","metric":"index points","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"xai","sourceTitle":"Grok 4.6 launch evaluations"},"score":61,"scoreUnit":"index","sourceUrl":"https://x.ai/news/grok-4-6"},{"benchmarkId":"aa-intelligence-index-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"aa-intelligence-index-source-release-snapshot-version-not-specified:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","id":"launch-xai-1321","ingestRunId":"launch-xai","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI.","family":"AA Intelligence Index","locator":"Performance table","metric":"index points","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"xai","sourceTitle":"Grok 4.6 launch evaluations"},"score":62,"scoreUnit":"index","sourceUrl":"https://x.ai/news/grok-4-6"},{"benchmarkId":"gdpval-aa-v2-elo-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gdpval-aa-v2-elo-source-release-snapshot-version-not-specified:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","id":"launch-xai-1322","ingestRunId":"launch-xai","modelId":"grok-4.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI.","family":"GDPval-AA v2 (Elo)","locator":"Performance table","metric":"Elo","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"xai","sourceTitle":"Grok 4.6 launch evaluations"},"score":1753,"scoreUnit":"elo","sourceUrl":"https://x.ai/news/grok-4-6"},{"benchmarkId":"gdpval-aa-v2-elo-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gdpval-aa-v2-elo-source-release-snapshot-version-not-specified:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","id":"launch-xai-1323","ingestRunId":"launch-xai","modelId":"grok-4.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI.","family":"GDPval-AA v2 (Elo)","locator":"Performance table","metric":"Elo","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"xai","sourceTitle":"Grok 4.6 launch evaluations"},"score":1526,"scoreUnit":"elo","sourceUrl":"https://x.ai/news/grok-4-6"},{"benchmarkId":"gdpval-aa-v2-elo-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gdpval-aa-v2-elo-source-release-snapshot-version-not-specified:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","id":"launch-xai-1324","ingestRunId":"launch-xai","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI.","family":"GDPval-AA v2 (Elo)","locator":"Performance table","metric":"Elo","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"xai","sourceTitle":"Grok 4.6 launch evaluations"},"score":1728,"scoreUnit":"elo","sourceUrl":"https://x.ai/news/grok-4-6"},{"benchmarkId":"gdpval-aa-v2-elo-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gdpval-aa-v2-elo-source-release-snapshot-version-not-specified:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","id":"launch-xai-1325","ingestRunId":"launch-xai","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI.","family":"GDPval-AA v2 (Elo)","locator":"Performance table","metric":"Elo","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"xai","sourceTitle":"Grok 4.6 launch evaluations"},"score":1741,"scoreUnit":"elo","sourceUrl":"https://x.ai/news/grok-4-6"},{"benchmarkId":"cursorbench-v3-2-3-2","evidenceKind":"lab_self_report","harnessId":"cursorbench-v3-2-3-2:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","id":"launch-xai-1326","ingestRunId":"launch-xai","modelId":"grok-4.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI.","family":"CursorBench v3.2","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"xai","sourceTitle":"Grok 4.6 launch evaluations"},"score":69.9,"scoreUnit":"percent","sourceUrl":"https://x.ai/news/grok-4-6"},{"benchmarkId":"cursorbench-v3-2-3-2","evidenceKind":"lab_self_report","harnessId":"cursorbench-v3-2-3-2:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","id":"launch-xai-1327","ingestRunId":"launch-xai","modelId":"grok-4.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI.","family":"CursorBench v3.2","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"xai","sourceTitle":"Grok 4.6 launch evaluations"},"score":66.7,"scoreUnit":"percent","sourceUrl":"https://x.ai/news/grok-4-6"},{"benchmarkId":"cursorbench-v3-2-3-2","evidenceKind":"lab_self_report","harnessId":"cursorbench-v3-2-3-2:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","id":"launch-xai-1328","ingestRunId":"launch-xai","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI.","family":"CursorBench v3.2","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"xai","sourceTitle":"Grok 4.6 launch evaluations"},"score":67.2,"scoreUnit":"percent","sourceUrl":"https://x.ai/news/grok-4-6"},{"benchmarkId":"cursorbench-v3-2-3-2","evidenceKind":"lab_self_report","harnessId":"cursorbench-v3-2-3-2:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","id":"launch-xai-1329","ingestRunId":"launch-xai","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI.","family":"CursorBench v3.2","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"xai","sourceTitle":"Grok 4.6 launch evaluations"},"score":70.5,"scoreUnit":"percent","sourceUrl":"https://x.ai/news/grok-4-6"},{"benchmarkId":"deepswe-v1-1-1-1-pct","evidenceKind":"lab_self_report","harnessId":"deepswe-v1-1-1-1-pct:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","id":"launch-xai-1330","ingestRunId":"launch-xai","modelId":"grok-4.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI.","family":"DeepSWE v1.1","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"xai","sourceTitle":"Grok 4.6 launch evaluations"},"score":65.9,"scoreUnit":"percent","sourceUrl":"https://x.ai/news/grok-4-6"},{"benchmarkId":"deepswe-v1-1-1-1-pct","evidenceKind":"lab_self_report","harnessId":"deepswe-v1-1-1-1-pct:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","id":"launch-xai-1331","ingestRunId":"launch-xai","modelId":"grok-4.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI.","family":"DeepSWE v1.1","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"xai","sourceTitle":"Grok 4.6 launch evaluations"},"score":54,"scoreUnit":"percent","sourceUrl":"https://x.ai/news/grok-4-6"},{"benchmarkId":"deepswe-v1-1-1-1-pct","evidenceKind":"lab_self_report","harnessId":"deepswe-v1-1-1-1-pct:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","id":"launch-xai-1332","ingestRunId":"launch-xai","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI.","family":"DeepSWE v1.1","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"xai","sourceTitle":"Grok 4.6 launch evaluations"},"score":73,"scoreUnit":"percent","sourceUrl":"https://x.ai/news/grok-4-6"},{"benchmarkId":"deepswe-v1-1-1-1-pct","evidenceKind":"lab_self_report","harnessId":"deepswe-v1-1-1-1-pct:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","id":"launch-xai-1333","ingestRunId":"launch-xai","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI.","family":"DeepSWE v1.1","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"xai","sourceTitle":"Grok 4.6 launch evaluations"},"score":70,"scoreUnit":"percent","sourceUrl":"https://x.ai/news/grok-4-6"},{"benchmarkId":"frontiercode-v1-1-extended-1-1","evidenceKind":"lab_self_report","harnessId":"frontiercode-v1-1-extended-1-1:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","id":"launch-xai-1334","ingestRunId":"launch-xai","modelId":"grok-4.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI.","family":"FrontierCode v1.1 Extended","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"xai","sourceTitle":"Grok 4.6 launch evaluations"},"score":61.3,"scoreUnit":"percent","sourceUrl":"https://x.ai/news/grok-4-6"},{"benchmarkId":"frontiercode-v1-1-extended-1-1","evidenceKind":"lab_self_report","harnessId":"frontiercode-v1-1-extended-1-1:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","id":"launch-xai-1335","ingestRunId":"launch-xai","modelId":"grok-4.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI.","family":"FrontierCode v1.1 Extended","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"xai","sourceTitle":"Grok 4.6 launch evaluations"},"score":56.6,"scoreUnit":"percent","sourceUrl":"https://x.ai/news/grok-4-6"},{"benchmarkId":"frontiercode-v1-1-extended-1-1","evidenceKind":"lab_self_report","harnessId":"frontiercode-v1-1-extended-1-1:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","id":"launch-xai-1336","ingestRunId":"launch-xai","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI.","family":"FrontierCode v1.1 Extended","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"xai","sourceTitle":"Grok 4.6 launch evaluations"},"score":60.6,"scoreUnit":"percent","sourceUrl":"https://x.ai/news/grok-4-6"},{"benchmarkId":"frontiercode-v1-1-extended-1-1","evidenceKind":"lab_self_report","harnessId":"frontiercode-v1-1-extended-1-1:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","id":"launch-xai-1337","ingestRunId":"launch-xai","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI.","family":"FrontierCode v1.1 Extended","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"xai","sourceTitle":"Grok 4.6 launch evaluations"},"score":63.6,"scoreUnit":"percent","sourceUrl":"https://x.ai/news/grok-4-6"},{"benchmarkId":"apex-agents-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"apex-agents-source-release-snapshot-version-not-specified:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","id":"launch-xai-1338","ingestRunId":"launch-xai","modelId":"grok-4.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI.","family":"APEX-Agents","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"xai","sourceTitle":"Grok 4.6 launch evaluations"},"score":57.5,"scoreUnit":"percent","sourceUrl":"https://x.ai/news/grok-4-6"},{"benchmarkId":"apex-agents-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"apex-agents-source-release-snapshot-version-not-specified:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","id":"launch-xai-1339","ingestRunId":"launch-xai","modelId":"grok-4.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI.","family":"APEX-Agents","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"xai","sourceTitle":"Grok 4.6 launch evaluations"},"score":47.1,"scoreUnit":"percent","sourceUrl":"https://x.ai/news/grok-4-6"},{"benchmarkId":"apex-agents-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"apex-agents-source-release-snapshot-version-not-specified:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","id":"launch-xai-1340","ingestRunId":"launch-xai","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI.","family":"APEX-Agents","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"xai","sourceTitle":"Grok 4.6 launch evaluations"},"score":56.7,"scoreUnit":"percent","sourceUrl":"https://x.ai/news/grok-4-6"},{"benchmarkId":"apex-agents-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"apex-agents-source-release-snapshot-version-not-specified:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","id":"launch-xai-1341","ingestRunId":"launch-xai","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI.","family":"APEX-Agents","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"xai","sourceTitle":"Grok 4.6 launch evaluations"},"score":59.2,"scoreUnit":"percent","sourceUrl":"https://x.ai/news/grok-4-6"},{"benchmarkId":"terminal-bench-v3-0-3-0","evidenceKind":"lab_self_report","harnessId":"terminal-bench-v3-0-3-0:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","id":"launch-xai-1342","ingestRunId":"launch-xai","modelId":"grok-4.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI.","family":"Terminal-Bench","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"xai","sourceTitle":"Grok 4.6 launch evaluations"},"score":26,"scoreUnit":"percent","sourceUrl":"https://x.ai/news/grok-4-6"},{"benchmarkId":"terminal-bench-v3-0-3-0","evidenceKind":"lab_self_report","harnessId":"terminal-bench-v3-0-3-0:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","id":"launch-xai-1343","ingestRunId":"launch-xai","modelId":"grok-4.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI.","family":"Terminal-Bench","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"xai","sourceTitle":"Grok 4.6 launch evaluations"},"score":15.7,"scoreUnit":"percent","sourceUrl":"https://x.ai/news/grok-4-6"},{"benchmarkId":"terminal-bench-v3-0-3-0","evidenceKind":"lab_self_report","harnessId":"terminal-bench-v3-0-3-0:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","id":"launch-xai-1344","ingestRunId":"launch-xai","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI.","family":"Terminal-Bench","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"xai","sourceTitle":"Grok 4.6 launch evaluations"},"score":34.6,"scoreUnit":"percent","sourceUrl":"https://x.ai/news/grok-4-6"},{"benchmarkId":"terminal-bench-v3-0-3-0","evidenceKind":"lab_self_report","harnessId":"terminal-bench-v3-0-3-0:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","id":"launch-xai-1345","ingestRunId":"launch-xai","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI.","family":"Terminal-Bench","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"xai","sourceTitle":"Grok 4.6 launch evaluations"},"score":34.1,"scoreUnit":"percent","sourceUrl":"https://x.ai/news/grok-4-6"},{"benchmarkId":"apex-swe-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"apex-swe-source-release-snapshot-version-not-specified:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","id":"launch-xai-1346","ingestRunId":"launch-xai","modelId":"grok-4.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI.","family":"APEX-SWE","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"xai","sourceTitle":"Grok 4.6 launch evaluations"},"score":56.4,"scoreUnit":"percent","sourceUrl":"https://x.ai/news/grok-4-6"},{"benchmarkId":"apex-swe-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"apex-swe-source-release-snapshot-version-not-specified:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","id":"launch-xai-1347","ingestRunId":"launch-xai","modelId":"grok-4.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI.","family":"APEX-SWE","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"xai","sourceTitle":"Grok 4.6 launch evaluations"},"score":53.6,"scoreUnit":"percent","sourceUrl":"https://x.ai/news/grok-4-6"},{"benchmarkId":"apex-swe-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"apex-swe-source-release-snapshot-version-not-specified:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","id":"launch-xai-1348","ingestRunId":"launch-xai","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"coding","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI.","family":"APEX-SWE","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"xai","sourceTitle":"Grok 4.6 launch evaluations"},"score":58.8,"scoreUnit":"percent","sourceUrl":"https://x.ai/news/grok-4-6"},{"benchmarkId":"aa-briefcase-elo-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"aa-briefcase-elo-source-release-snapshot-version-not-specified:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","id":"launch-xai-1349","ingestRunId":"launch-xai","modelId":"grok-4.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI.","family":"AA-Briefcase (Elo)","locator":"Performance table","metric":"Elo","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"xai","sourceTitle":"Grok 4.6 launch evaluations"},"score":1577,"scoreUnit":"elo","sourceUrl":"https://x.ai/news/grok-4-6"},{"benchmarkId":"aa-briefcase-elo-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"aa-briefcase-elo-source-release-snapshot-version-not-specified:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","id":"launch-xai-1350","ingestRunId":"launch-xai","modelId":"grok-4.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI.","family":"AA-Briefcase (Elo)","locator":"Performance table","metric":"Elo","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"xai","sourceTitle":"Grok 4.6 launch evaluations"},"score":1313,"scoreUnit":"elo","sourceUrl":"https://x.ai/news/grok-4-6"},{"benchmarkId":"aa-briefcase-elo-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"aa-briefcase-elo-source-release-snapshot-version-not-specified:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","id":"launch-xai-1351","ingestRunId":"launch-xai","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI.","family":"AA-Briefcase (Elo)","locator":"Performance table","metric":"Elo","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"xai","sourceTitle":"Grok 4.6 launch evaluations"},"score":1502,"scoreUnit":"elo","sourceUrl":"https://x.ai/news/grok-4-6"},{"benchmarkId":"aa-briefcase-elo-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"aa-briefcase-elo-source-release-snapshot-version-not-specified:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","id":"launch-xai-1352","ingestRunId":"launch-xai","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI.","family":"AA-Briefcase (Elo)","locator":"Performance table","metric":"Elo","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"xai","sourceTitle":"Grok 4.6 launch evaluations"},"score":1574,"scoreUnit":"elo","sourceUrl":"https://x.ai/news/grok-4-6"},{"benchmarkId":"harvey-lab-vals-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"harvey-lab-vals-source-release-snapshot-version-not-specified:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","id":"launch-xai-1353","ingestRunId":"launch-xai","modelId":"grok-4.6","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI.","family":"Harvey LAB (Vals)","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"xai","sourceTitle":"Grok 4.6 launch evaluations"},"score":15.8,"scoreUnit":"percent","sourceUrl":"https://x.ai/news/grok-4-6"},{"benchmarkId":"harvey-lab-vals-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"harvey-lab-vals-source-release-snapshot-version-not-specified:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","id":"launch-xai-1354","ingestRunId":"launch-xai","modelId":"grok-4.5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI.","family":"Harvey LAB (Vals)","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"xai","sourceTitle":"Grok 4.6 launch evaluations"},"score":12.9,"scoreUnit":"percent","sourceUrl":"https://x.ai/news/grok-4-6"},{"benchmarkId":"harvey-lab-vals-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"harvey-lab-vals-source-release-snapshot-version-not-specified:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","id":"launch-xai-1355","ingestRunId":"launch-xai","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI.","family":"Harvey LAB (Vals)","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"xai","sourceTitle":"Grok 4.6 launch evaluations"},"score":2.5,"scoreUnit":"percent","sourceUrl":"https://x.ai/news/grok-4-6"},{"benchmarkId":"harvey-lab-vals-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"harvey-lab-vals-source-release-snapshot-version-not-specified:xai:R3JvayBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IEZhYmxlIDUgTWF4LiBDb21wZXRpdG9yIGZpZ3VyZXMgZnJvbSBkZXZlbG9wZXIgY2FyZHMvcHVibGljIGxlYWRlcmJvYXJkcyBhcyBzZWxlY3RlZCBieSB4QUku","id":"launch-xai-1356","ingestRunId":"launch-xai","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-06","rawPayloadHash":null,"report":{"category":"agentic","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI.","family":"Harvey LAB (Vals)","locator":"Performance table","metric":"%","notes":"First-party reported result; comparator results retain the source evaluation setup.","sourceId":"xai","sourceTitle":"Grok 4.6 launch evaluations"},"score":11.3,"scoreUnit":"percent","sourceUrl":"https://x.ai/news/grok-4-6"},{"benchmarkId":"cursorbench-4-0","evidenceKind":"lab_self_report","harnessId":"cursorbench-4-0:xai-grok47:R3JvayA0LjcgWEhpZ2ggKERlZXBTV0U6IEhpZ2gpOyBHcm9rIDQuNiBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IENsYXVkZSBGYWJsZSA1LjEgTWF4LiBMYXVuY2ggY29tcGFyaXNvbjsgcGVyLW1vZGVsIGFnZW50IGhhcm5lc3MsIGV2YWx1YXRpb24gYnVkZ2V0IGFuZCByZXJ1biBwcm92ZW5hbmNlIGFyZSBub3Qgc3BlY2lmaWVkLg","id":"launch-xai-grok47-2230","ingestRunId":"launch-xai-grok47","modelId":"grok-4.7","n":null,"observedOn":"2026-09-21","rawPayloadHash":"2d9201e3d2f6a14d0a636ed6d0c8aa5b7b3da8c25dccbd911c7d7a2364a66def","report":{"category":"coding","comparabilityReason":"The launch table does not establish matching agent harnesses and evaluation settings. Retained as a provider claim; not a new comparable evaluation.","comparable":false,"configuration":"Grok 4.7 XHigh (DeepSWE: High); Grok 4.6 High; GPT-5.6 Sol Max; Claude Fable 5.1 Max. Launch comparison; per-model agent harness, evaluation budget and rerun provenance are not specified.","family":"CursorBench","locator":"Model Improvements table: CursorBench","metric":"%","notes":"Official launch table. Effort follows the column header except the Grok 4.7 DeepSWE asterisk, which explicitly means High.","sourceId":"xai-grok47","sourceTitle":"Grok 4.7 launch comparison"},"score":46.3,"scoreUnit":"percent","sourceUrl":"https://x.ai/news/grok-4-7"},{"benchmarkId":"cursorbench-4-0","evidenceKind":"lab_self_report","harnessId":"cursorbench-4-0:xai-grok47:R3JvayA0LjcgWEhpZ2ggKERlZXBTV0U6IEhpZ2gpOyBHcm9rIDQuNiBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IENsYXVkZSBGYWJsZSA1LjEgTWF4LiBMYXVuY2ggY29tcGFyaXNvbjsgcGVyLW1vZGVsIGFnZW50IGhhcm5lc3MsIGV2YWx1YXRpb24gYnVkZ2V0IGFuZCByZXJ1biBwcm92ZW5hbmNlIGFyZSBub3Qgc3BlY2lmaWVkLg","id":"launch-xai-grok47-2231","ingestRunId":"launch-xai-grok47","modelId":"grok-4.6","n":null,"observedOn":"2026-09-21","rawPayloadHash":"2d9201e3d2f6a14d0a636ed6d0c8aa5b7b3da8c25dccbd911c7d7a2364a66def","report":{"category":"coding","comparabilityReason":"The launch table does not establish matching agent harnesses and evaluation settings. Retained as a provider claim; not a new comparable evaluation.","comparable":false,"configuration":"Grok 4.7 XHigh (DeepSWE: High); Grok 4.6 High; GPT-5.6 Sol Max; Claude Fable 5.1 Max. Launch comparison; per-model agent harness, evaluation budget and rerun provenance are not specified.","family":"CursorBench","locator":"Model Improvements table: CursorBench","metric":"%","notes":"Official launch table. Effort follows the column header except the Grok 4.7 DeepSWE asterisk, which explicitly means High.","sourceId":"xai-grok47","sourceTitle":"Grok 4.7 launch comparison"},"score":40.4,"scoreUnit":"percent","sourceUrl":"https://x.ai/news/grok-4-7"},{"benchmarkId":"cursorbench-4-0","evidenceKind":"lab_self_report","harnessId":"cursorbench-4-0:xai-grok47:R3JvayA0LjcgWEhpZ2ggKERlZXBTV0U6IEhpZ2gpOyBHcm9rIDQuNiBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IENsYXVkZSBGYWJsZSA1LjEgTWF4LiBMYXVuY2ggY29tcGFyaXNvbjsgcGVyLW1vZGVsIGFnZW50IGhhcm5lc3MsIGV2YWx1YXRpb24gYnVkZ2V0IGFuZCByZXJ1biBwcm92ZW5hbmNlIGFyZSBub3Qgc3BlY2lmaWVkLg","id":"launch-xai-grok47-2232","ingestRunId":"launch-xai-grok47","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-21","rawPayloadHash":"2d9201e3d2f6a14d0a636ed6d0c8aa5b7b3da8c25dccbd911c7d7a2364a66def","report":{"category":"coding","comparabilityReason":"The launch table does not establish matching agent harnesses and evaluation settings. Retained as a provider claim; not a new comparable evaluation.","comparable":false,"configuration":"Grok 4.7 XHigh (DeepSWE: High); Grok 4.6 High; GPT-5.6 Sol Max; Claude Fable 5.1 Max. Launch comparison; per-model agent harness, evaluation budget and rerun provenance are not specified.","family":"CursorBench","locator":"Model Improvements table: CursorBench","metric":"%","notes":"Official launch table. Effort follows the column header except the Grok 4.7 DeepSWE asterisk, which explicitly means High.","sourceId":"xai-grok47","sourceTitle":"Grok 4.7 launch comparison"},"score":41.7,"scoreUnit":"percent","sourceUrl":"https://x.ai/news/grok-4-7"},{"benchmarkId":"cursorbench-4-0","evidenceKind":"lab_self_report","harnessId":"cursorbench-4-0:xai-grok47:R3JvayA0LjcgWEhpZ2ggKERlZXBTV0U6IEhpZ2gpOyBHcm9rIDQuNiBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IENsYXVkZSBGYWJsZSA1LjEgTWF4LiBMYXVuY2ggY29tcGFyaXNvbjsgcGVyLW1vZGVsIGFnZW50IGhhcm5lc3MsIGV2YWx1YXRpb24gYnVkZ2V0IGFuZCByZXJ1biBwcm92ZW5hbmNlIGFyZSBub3Qgc3BlY2lmaWVkLg","id":"launch-xai-grok47-2233","ingestRunId":"launch-xai-grok47","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-21","rawPayloadHash":"2d9201e3d2f6a14d0a636ed6d0c8aa5b7b3da8c25dccbd911c7d7a2364a66def","report":{"category":"coding","comparabilityReason":"The launch table does not establish matching agent harnesses and evaluation settings. Retained as a provider claim; not a new comparable evaluation.","comparable":false,"configuration":"Grok 4.7 XHigh (DeepSWE: High); Grok 4.6 High; GPT-5.6 Sol Max; Claude Fable 5.1 Max. Launch comparison; per-model agent harness, evaluation budget and rerun provenance are not specified.","family":"CursorBench","locator":"Model Improvements table: CursorBench","metric":"%","notes":"Official launch table. Effort follows the column header except the Grok 4.7 DeepSWE asterisk, which explicitly means High.","sourceId":"xai-grok47","sourceTitle":"Grok 4.7 launch comparison"},"score":51.8,"scoreUnit":"percent","sourceUrl":"https://x.ai/news/grok-4-7"},{"benchmarkId":"deepswe-v1-1-1-1","evidenceKind":"lab_self_report","harnessId":"deepswe-v1-1-1-1:xai-grok47:R3JvayA0LjcgWEhpZ2ggKERlZXBTV0U6IEhpZ2gpOyBHcm9rIDQuNiBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IENsYXVkZSBGYWJsZSA1LjEgTWF4LiBMYXVuY2ggY29tcGFyaXNvbjsgcGVyLW1vZGVsIGFnZW50IGhhcm5lc3MsIGV2YWx1YXRpb24gYnVkZ2V0IGFuZCByZXJ1biBwcm92ZW5hbmNlIGFyZSBub3Qgc3BlY2lmaWVkLg","id":"launch-xai-grok47-2234","ingestRunId":"launch-xai-grok47","modelId":"grok-4.7","n":null,"observedOn":"2026-09-21","rawPayloadHash":"2d9201e3d2f6a14d0a636ed6d0c8aa5b7b3da8c25dccbd911c7d7a2364a66def","report":{"category":"coding","comparabilityReason":"The launch table does not establish matching agent harnesses and evaluation settings. Retained as a provider claim; not a new comparable evaluation.","comparable":false,"configuration":"Grok 4.7 XHigh (DeepSWE: High); Grok 4.6 High; GPT-5.6 Sol Max; Claude Fable 5.1 Max. Launch comparison; per-model agent harness, evaluation budget and rerun provenance are not specified.","family":"DeepSWE v1.1","locator":"Model Improvements table: DeepSWE v1.1","metric":"%","notes":"Official launch table. Effort follows the column header except the Grok 4.7 DeepSWE asterisk, which explicitly means High.","sourceId":"xai-grok47","sourceTitle":"Grok 4.7 launch comparison"},"score":71,"scoreUnit":"percent","sourceUrl":"https://x.ai/news/grok-4-7"},{"benchmarkId":"deepswe-v1-1-1-1","evidenceKind":"lab_self_report","harnessId":"deepswe-v1-1-1-1:xai-grok47:R3JvayA0LjcgWEhpZ2ggKERlZXBTV0U6IEhpZ2gpOyBHcm9rIDQuNiBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IENsYXVkZSBGYWJsZSA1LjEgTWF4LiBMYXVuY2ggY29tcGFyaXNvbjsgcGVyLW1vZGVsIGFnZW50IGhhcm5lc3MsIGV2YWx1YXRpb24gYnVkZ2V0IGFuZCByZXJ1biBwcm92ZW5hbmNlIGFyZSBub3Qgc3BlY2lmaWVkLg","id":"launch-xai-grok47-2235","ingestRunId":"launch-xai-grok47","modelId":"grok-4.6","n":null,"observedOn":"2026-09-21","rawPayloadHash":"2d9201e3d2f6a14d0a636ed6d0c8aa5b7b3da8c25dccbd911c7d7a2364a66def","report":{"category":"coding","comparabilityReason":"The launch table does not establish matching agent harnesses and evaluation settings. Retained as a provider claim; not a new comparable evaluation.","comparable":false,"configuration":"Grok 4.7 XHigh (DeepSWE: High); Grok 4.6 High; GPT-5.6 Sol Max; Claude Fable 5.1 Max. Launch comparison; per-model agent harness, evaluation budget and rerun provenance are not specified.","family":"DeepSWE v1.1","locator":"Model Improvements table: DeepSWE v1.1","metric":"%","notes":"Official launch table. Effort follows the column header except the Grok 4.7 DeepSWE asterisk, which explicitly means High.","sourceId":"xai-grok47","sourceTitle":"Grok 4.7 launch comparison"},"score":65.2,"scoreUnit":"percent","sourceUrl":"https://x.ai/news/grok-4-7"},{"benchmarkId":"deepswe-v1-1-1-1","evidenceKind":"lab_self_report","harnessId":"deepswe-v1-1-1-1:xai-grok47:R3JvayA0LjcgWEhpZ2ggKERlZXBTV0U6IEhpZ2gpOyBHcm9rIDQuNiBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IENsYXVkZSBGYWJsZSA1LjEgTWF4LiBMYXVuY2ggY29tcGFyaXNvbjsgcGVyLW1vZGVsIGFnZW50IGhhcm5lc3MsIGV2YWx1YXRpb24gYnVkZ2V0IGFuZCByZXJ1biBwcm92ZW5hbmNlIGFyZSBub3Qgc3BlY2lmaWVkLg","id":"launch-xai-grok47-2236","ingestRunId":"launch-xai-grok47","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-21","rawPayloadHash":"2d9201e3d2f6a14d0a636ed6d0c8aa5b7b3da8c25dccbd911c7d7a2364a66def","report":{"category":"coding","comparabilityReason":"The launch table does not establish matching agent harnesses and evaluation settings. Retained as a provider claim; not a new comparable evaluation.","comparable":false,"configuration":"Grok 4.7 XHigh (DeepSWE: High); Grok 4.6 High; GPT-5.6 Sol Max; Claude Fable 5.1 Max. Launch comparison; per-model agent harness, evaluation budget and rerun provenance are not specified.","family":"DeepSWE v1.1","locator":"Model Improvements table: DeepSWE v1.1","metric":"%","notes":"Official launch table. Effort follows the column header except the Grok 4.7 DeepSWE asterisk, which explicitly means High.","sourceId":"xai-grok47","sourceTitle":"Grok 4.7 launch comparison"},"score":72.7,"scoreUnit":"percent","sourceUrl":"https://x.ai/news/grok-4-7"},{"benchmarkId":"deepswe-v1-1-1-1","evidenceKind":"lab_self_report","harnessId":"deepswe-v1-1-1-1:xai-grok47:R3JvayA0LjcgWEhpZ2ggKERlZXBTV0U6IEhpZ2gpOyBHcm9rIDQuNiBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IENsYXVkZSBGYWJsZSA1LjEgTWF4LiBMYXVuY2ggY29tcGFyaXNvbjsgcGVyLW1vZGVsIGFnZW50IGhhcm5lc3MsIGV2YWx1YXRpb24gYnVkZ2V0IGFuZCByZXJ1biBwcm92ZW5hbmNlIGFyZSBub3Qgc3BlY2lmaWVkLg","id":"launch-xai-grok47-2237","ingestRunId":"launch-xai-grok47","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-21","rawPayloadHash":"2d9201e3d2f6a14d0a636ed6d0c8aa5b7b3da8c25dccbd911c7d7a2364a66def","report":{"category":"coding","comparabilityReason":"The launch table does not establish matching agent harnesses and evaluation settings. Retained as a provider claim; not a new comparable evaluation.","comparable":false,"configuration":"Grok 4.7 XHigh (DeepSWE: High); Grok 4.6 High; GPT-5.6 Sol Max; Claude Fable 5.1 Max. Launch comparison; per-model agent harness, evaluation budget and rerun provenance are not specified.","family":"DeepSWE v1.1","locator":"Model Improvements table: DeepSWE v1.1","metric":"%","notes":"Official launch table. Effort follows the column header except the Grok 4.7 DeepSWE asterisk, which explicitly means High.","sourceId":"xai-grok47","sourceTitle":"Grok 4.7 launch comparison"},"score":70,"scoreUnit":"percent","sourceUrl":"https://x.ai/news/grok-4-7"},{"benchmarkId":"eebench","evidenceKind":"lab_self_report","harnessId":"eebench:xai-grok47:R3JvayA0LjcgWEhpZ2ggKERlZXBTV0U6IEhpZ2gpOyBHcm9rIDQuNiBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IENsYXVkZSBGYWJsZSA1LjEgTWF4LiBMYXVuY2ggY29tcGFyaXNvbjsgcGVyLW1vZGVsIGFnZW50IGhhcm5lc3MsIGV2YWx1YXRpb24gYnVkZ2V0IGFuZCByZXJ1biBwcm92ZW5hbmNlIGFyZSBub3Qgc3BlY2lmaWVkLg","id":"launch-xai-grok47-2238","ingestRunId":"launch-xai-grok47","modelId":"grok-4.7","n":null,"observedOn":"2026-09-21","rawPayloadHash":"2d9201e3d2f6a14d0a636ed6d0c8aa5b7b3da8c25dccbd911c7d7a2364a66def","report":{"category":"agentic","comparabilityReason":"The launch table does not establish matching agent harnesses and evaluation settings. Retained as a provider claim; not a new comparable evaluation.","comparable":false,"configuration":"Grok 4.7 XHigh (DeepSWE: High); Grok 4.6 High; GPT-5.6 Sol Max; Claude Fable 5.1 Max. Launch comparison; per-model agent harness, evaluation budget and rerun provenance are not specified.","family":"EEBench","locator":"Model Improvements table: EEBench","metric":"%","notes":"Official launch table. Effort follows the column header except the Grok 4.7 DeepSWE asterisk, which explicitly means High.","sourceId":"xai-grok47","sourceTitle":"Grok 4.7 launch comparison"},"score":64,"scoreUnit":"percent","sourceUrl":"https://x.ai/news/grok-4-7"},{"benchmarkId":"eebench","evidenceKind":"lab_self_report","harnessId":"eebench:xai-grok47:R3JvayA0LjcgWEhpZ2ggKERlZXBTV0U6IEhpZ2gpOyBHcm9rIDQuNiBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IENsYXVkZSBGYWJsZSA1LjEgTWF4LiBMYXVuY2ggY29tcGFyaXNvbjsgcGVyLW1vZGVsIGFnZW50IGhhcm5lc3MsIGV2YWx1YXRpb24gYnVkZ2V0IGFuZCByZXJ1biBwcm92ZW5hbmNlIGFyZSBub3Qgc3BlY2lmaWVkLg","id":"launch-xai-grok47-2239","ingestRunId":"launch-xai-grok47","modelId":"grok-4.6","n":null,"observedOn":"2026-09-21","rawPayloadHash":"2d9201e3d2f6a14d0a636ed6d0c8aa5b7b3da8c25dccbd911c7d7a2364a66def","report":{"category":"agentic","comparabilityReason":"The launch table does not establish matching agent harnesses and evaluation settings. Retained as a provider claim; not a new comparable evaluation.","comparable":false,"configuration":"Grok 4.7 XHigh (DeepSWE: High); Grok 4.6 High; GPT-5.6 Sol Max; Claude Fable 5.1 Max. Launch comparison; per-model agent harness, evaluation budget and rerun provenance are not specified.","family":"EEBench","locator":"Model Improvements table: EEBench","metric":"%","notes":"Official launch table. Effort follows the column header except the Grok 4.7 DeepSWE asterisk, which explicitly means High.","sourceId":"xai-grok47","sourceTitle":"Grok 4.7 launch comparison"},"score":53,"scoreUnit":"percent","sourceUrl":"https://x.ai/news/grok-4-7"},{"benchmarkId":"eebench","evidenceKind":"lab_self_report","harnessId":"eebench:xai-grok47:R3JvayA0LjcgWEhpZ2ggKERlZXBTV0U6IEhpZ2gpOyBHcm9rIDQuNiBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IENsYXVkZSBGYWJsZSA1LjEgTWF4LiBMYXVuY2ggY29tcGFyaXNvbjsgcGVyLW1vZGVsIGFnZW50IGhhcm5lc3MsIGV2YWx1YXRpb24gYnVkZ2V0IGFuZCByZXJ1biBwcm92ZW5hbmNlIGFyZSBub3Qgc3BlY2lmaWVkLg","id":"launch-xai-grok47-2240","ingestRunId":"launch-xai-grok47","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-21","rawPayloadHash":"2d9201e3d2f6a14d0a636ed6d0c8aa5b7b3da8c25dccbd911c7d7a2364a66def","report":{"category":"agentic","comparabilityReason":"The launch table does not establish matching agent harnesses and evaluation settings. Retained as a provider claim; not a new comparable evaluation.","comparable":false,"configuration":"Grok 4.7 XHigh (DeepSWE: High); Grok 4.6 High; GPT-5.6 Sol Max; Claude Fable 5.1 Max. Launch comparison; per-model agent harness, evaluation budget and rerun provenance are not specified.","family":"EEBench","locator":"Model Improvements table: EEBench","metric":"%","notes":"Official launch table. Effort follows the column header except the Grok 4.7 DeepSWE asterisk, which explicitly means High.","sourceId":"xai-grok47","sourceTitle":"Grok 4.7 launch comparison"},"score":39.4,"scoreUnit":"percent","sourceUrl":"https://x.ai/news/grok-4-7"},{"benchmarkId":"eebench","evidenceKind":"lab_self_report","harnessId":"eebench:xai-grok47:R3JvayA0LjcgWEhpZ2ggKERlZXBTV0U6IEhpZ2gpOyBHcm9rIDQuNiBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IENsYXVkZSBGYWJsZSA1LjEgTWF4LiBMYXVuY2ggY29tcGFyaXNvbjsgcGVyLW1vZGVsIGFnZW50IGhhcm5lc3MsIGV2YWx1YXRpb24gYnVkZ2V0IGFuZCByZXJ1biBwcm92ZW5hbmNlIGFyZSBub3Qgc3BlY2lmaWVkLg","id":"launch-xai-grok47-2241","ingestRunId":"launch-xai-grok47","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-21","rawPayloadHash":"2d9201e3d2f6a14d0a636ed6d0c8aa5b7b3da8c25dccbd911c7d7a2364a66def","report":{"category":"agentic","comparabilityReason":"The launch table does not establish matching agent harnesses and evaluation settings. Retained as a provider claim; not a new comparable evaluation.","comparable":false,"configuration":"Grok 4.7 XHigh (DeepSWE: High); Grok 4.6 High; GPT-5.6 Sol Max; Claude Fable 5.1 Max. Launch comparison; per-model agent harness, evaluation budget and rerun provenance are not specified.","family":"EEBench","locator":"Model Improvements table: EEBench","metric":"%","notes":"Official launch table. Effort follows the column header except the Grok 4.7 DeepSWE asterisk, which explicitly means High.","sourceId":"xai-grok47","sourceTitle":"Grok 4.7 launch comparison"},"score":56.4,"scoreUnit":"percent","sourceUrl":"https://x.ai/news/grok-4-7"},{"benchmarkId":"aa-briefcase-1-1","evidenceKind":"lab_self_report","harnessId":"aa-briefcase-1-1:xai-grok47:R3JvayA0LjcgWEhpZ2ggKERlZXBTV0U6IEhpZ2gpOyBHcm9rIDQuNiBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IENsYXVkZSBGYWJsZSA1LjEgTWF4LiBMYXVuY2ggY29tcGFyaXNvbjsgcGVyLW1vZGVsIGFnZW50IGhhcm5lc3MsIGV2YWx1YXRpb24gYnVkZ2V0IGFuZCByZXJ1biBwcm92ZW5hbmNlIGFyZSBub3Qgc3BlY2lmaWVkLg","id":"launch-xai-grok47-2242","ingestRunId":"launch-xai-grok47","modelId":"grok-4.7","n":null,"observedOn":"2026-09-21","rawPayloadHash":"2d9201e3d2f6a14d0a636ed6d0c8aa5b7b3da8c25dccbd911c7d7a2364a66def","report":{"category":"agentic","comparabilityReason":"Secondary Artificial Analysis leaderboard report; not an additional independent evaluation. Use the original evaluator for reviewed exact-configuration comparisons.","comparable":false,"configuration":"Grok 4.7 XHigh (DeepSWE: High); Grok 4.6 High; GPT-5.6 Sol Max; Claude Fable 5.1 Max. Launch comparison; per-model agent harness, evaluation budget and rerun provenance are not specified.","family":"AA Briefcase","locator":"Model Improvements table: AA Briefcase","metric":"Elo","notes":"Official launch table. Effort follows the column header except the Grok 4.7 DeepSWE asterisk, which explicitly means High. Original evaluator: https://artificialanalysis.ai/evaluations/aa-briefcase","sourceId":"xai-grok47","sourceTitle":"Grok 4.7 launch comparison"},"score":1657,"scoreUnit":"elo","sourceUrl":"https://x.ai/news/grok-4-7"},{"benchmarkId":"aa-briefcase-1-1","evidenceKind":"lab_self_report","harnessId":"aa-briefcase-1-1:xai-grok47:R3JvayA0LjcgWEhpZ2ggKERlZXBTV0U6IEhpZ2gpOyBHcm9rIDQuNiBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IENsYXVkZSBGYWJsZSA1LjEgTWF4LiBMYXVuY2ggY29tcGFyaXNvbjsgcGVyLW1vZGVsIGFnZW50IGhhcm5lc3MsIGV2YWx1YXRpb24gYnVkZ2V0IGFuZCByZXJ1biBwcm92ZW5hbmNlIGFyZSBub3Qgc3BlY2lmaWVkLg","id":"launch-xai-grok47-2243","ingestRunId":"launch-xai-grok47","modelId":"grok-4.6","n":null,"observedOn":"2026-09-21","rawPayloadHash":"2d9201e3d2f6a14d0a636ed6d0c8aa5b7b3da8c25dccbd911c7d7a2364a66def","report":{"category":"agentic","comparabilityReason":"Secondary Artificial Analysis leaderboard report; not an additional independent evaluation. Use the original evaluator for reviewed exact-configuration comparisons.","comparable":false,"configuration":"Grok 4.7 XHigh (DeepSWE: High); Grok 4.6 High; GPT-5.6 Sol Max; Claude Fable 5.1 Max. Launch comparison; per-model agent harness, evaluation budget and rerun provenance are not specified.","family":"AA Briefcase","locator":"Model Improvements table: AA Briefcase","metric":"Elo","notes":"Official launch table. Effort follows the column header except the Grok 4.7 DeepSWE asterisk, which explicitly means High. Original evaluator: https://artificialanalysis.ai/evaluations/aa-briefcase","sourceId":"xai-grok47","sourceTitle":"Grok 4.7 launch comparison"},"score":1546,"scoreUnit":"elo","sourceUrl":"https://x.ai/news/grok-4-7"},{"benchmarkId":"aa-briefcase-1-1","evidenceKind":"lab_self_report","harnessId":"aa-briefcase-1-1:xai-grok47:R3JvayA0LjcgWEhpZ2ggKERlZXBTV0U6IEhpZ2gpOyBHcm9rIDQuNiBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IENsYXVkZSBGYWJsZSA1LjEgTWF4LiBMYXVuY2ggY29tcGFyaXNvbjsgcGVyLW1vZGVsIGFnZW50IGhhcm5lc3MsIGV2YWx1YXRpb24gYnVkZ2V0IGFuZCByZXJ1biBwcm92ZW5hbmNlIGFyZSBub3Qgc3BlY2lmaWVkLg","id":"launch-xai-grok47-2244","ingestRunId":"launch-xai-grok47","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-21","rawPayloadHash":"2d9201e3d2f6a14d0a636ed6d0c8aa5b7b3da8c25dccbd911c7d7a2364a66def","report":{"category":"agentic","comparabilityReason":"Secondary Artificial Analysis leaderboard report; not an additional independent evaluation. Use the original evaluator for reviewed exact-configuration comparisons.","comparable":false,"configuration":"Grok 4.7 XHigh (DeepSWE: High); Grok 4.6 High; GPT-5.6 Sol Max; Claude Fable 5.1 Max. Launch comparison; per-model agent harness, evaluation budget and rerun provenance are not specified.","family":"AA Briefcase","locator":"Model Improvements table: AA Briefcase","metric":"Elo","notes":"Official launch table. Effort follows the column header except the Grok 4.7 DeepSWE asterisk, which explicitly means High. Original evaluator: https://artificialanalysis.ai/evaluations/aa-briefcase","sourceId":"xai-grok47","sourceTitle":"Grok 4.7 launch comparison"},"score":1487,"scoreUnit":"elo","sourceUrl":"https://x.ai/news/grok-4-7"},{"benchmarkId":"aa-briefcase-1-1","evidenceKind":"lab_self_report","harnessId":"aa-briefcase-1-1:xai-grok47:R3JvayA0LjcgWEhpZ2ggKERlZXBTV0U6IEhpZ2gpOyBHcm9rIDQuNiBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IENsYXVkZSBGYWJsZSA1LjEgTWF4LiBMYXVuY2ggY29tcGFyaXNvbjsgcGVyLW1vZGVsIGFnZW50IGhhcm5lc3MsIGV2YWx1YXRpb24gYnVkZ2V0IGFuZCByZXJ1biBwcm92ZW5hbmNlIGFyZSBub3Qgc3BlY2lmaWVkLg","id":"launch-xai-grok47-2245","ingestRunId":"launch-xai-grok47","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-21","rawPayloadHash":"2d9201e3d2f6a14d0a636ed6d0c8aa5b7b3da8c25dccbd911c7d7a2364a66def","report":{"category":"agentic","comparabilityReason":"Secondary Artificial Analysis leaderboard report; not an additional independent evaluation. Use the original evaluator for reviewed exact-configuration comparisons.","comparable":false,"configuration":"Grok 4.7 XHigh (DeepSWE: High); Grok 4.6 High; GPT-5.6 Sol Max; Claude Fable 5.1 Max. Launch comparison; per-model agent harness, evaluation budget and rerun provenance are not specified.","family":"AA Briefcase","locator":"Model Improvements table: AA Briefcase","metric":"Elo","notes":"Official launch table. Effort follows the column header except the Grok 4.7 DeepSWE asterisk, which explicitly means High. Original evaluator: https://artificialanalysis.ai/evaluations/aa-briefcase","sourceId":"xai-grok47","sourceTitle":"Grok 4.7 launch comparison"},"score":1678,"scoreUnit":"elo","sourceUrl":"https://x.ai/news/grok-4-7"},{"benchmarkId":"terminal-bench-4-0-pct","evidenceKind":"lab_self_report","harnessId":"terminal-bench-4-0-pct:xai-grok47:R3JvayA0LjcgWEhpZ2ggKERlZXBTV0U6IEhpZ2gpOyBHcm9rIDQuNiBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IENsYXVkZSBGYWJsZSA1LjEgTWF4LiBMYXVuY2ggY29tcGFyaXNvbjsgcGVyLW1vZGVsIGFnZW50IGhhcm5lc3MsIGV2YWx1YXRpb24gYnVkZ2V0IGFuZCByZXJ1biBwcm92ZW5hbmNlIGFyZSBub3Qgc3BlY2lmaWVkLg","id":"launch-xai-grok47-2246","ingestRunId":"launch-xai-grok47","modelId":"grok-4.7","n":null,"observedOn":"2026-09-21","rawPayloadHash":"2d9201e3d2f6a14d0a636ed6d0c8aa5b7b3da8c25dccbd911c7d7a2364a66def","report":{"category":"coding","comparabilityReason":"The launch table does not establish matching agent harnesses and evaluation settings. Retained as a provider claim; not a new comparable evaluation.","comparable":false,"configuration":"Grok 4.7 XHigh (DeepSWE: High); Grok 4.6 High; GPT-5.6 Sol Max; Claude Fable 5.1 Max. Launch comparison; per-model agent harness, evaluation budget and rerun provenance are not specified.","family":"Terminal-Bench","locator":"Model Improvements table: Terminal-Bench","metric":"%","notes":"Official launch table. Effort follows the column header except the Grok 4.7 DeepSWE asterisk, which explicitly means High.","sourceId":"xai-grok47","sourceTitle":"Grok 4.7 launch comparison"},"score":38,"scoreUnit":"percent","sourceUrl":"https://x.ai/news/grok-4-7"},{"benchmarkId":"terminal-bench-4-0-pct","evidenceKind":"lab_self_report","harnessId":"terminal-bench-4-0-pct:xai-grok47:R3JvayA0LjcgWEhpZ2ggKERlZXBTV0U6IEhpZ2gpOyBHcm9rIDQuNiBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IENsYXVkZSBGYWJsZSA1LjEgTWF4LiBMYXVuY2ggY29tcGFyaXNvbjsgcGVyLW1vZGVsIGFnZW50IGhhcm5lc3MsIGV2YWx1YXRpb24gYnVkZ2V0IGFuZCByZXJ1biBwcm92ZW5hbmNlIGFyZSBub3Qgc3BlY2lmaWVkLg","id":"launch-xai-grok47-2247","ingestRunId":"launch-xai-grok47","modelId":"grok-4.6","n":null,"observedOn":"2026-09-21","rawPayloadHash":"2d9201e3d2f6a14d0a636ed6d0c8aa5b7b3da8c25dccbd911c7d7a2364a66def","report":{"category":"coding","comparabilityReason":"The launch table does not establish matching agent harnesses and evaluation settings. Retained as a provider claim; not a new comparable evaluation.","comparable":false,"configuration":"Grok 4.7 XHigh (DeepSWE: High); Grok 4.6 High; GPT-5.6 Sol Max; Claude Fable 5.1 Max. Launch comparison; per-model agent harness, evaluation budget and rerun provenance are not specified.","family":"Terminal-Bench","locator":"Model Improvements table: Terminal-Bench","metric":"%","notes":"Official launch table. Effort follows the column header except the Grok 4.7 DeepSWE asterisk, which explicitly means High.","sourceId":"xai-grok47","sourceTitle":"Grok 4.7 launch comparison"},"score":20.3,"scoreUnit":"percent","sourceUrl":"https://x.ai/news/grok-4-7"},{"benchmarkId":"terminal-bench-4-0-pct","evidenceKind":"lab_self_report","harnessId":"terminal-bench-4-0-pct:xai-grok47:R3JvayA0LjcgWEhpZ2ggKERlZXBTV0U6IEhpZ2gpOyBHcm9rIDQuNiBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IENsYXVkZSBGYWJsZSA1LjEgTWF4LiBMYXVuY2ggY29tcGFyaXNvbjsgcGVyLW1vZGVsIGFnZW50IGhhcm5lc3MsIGV2YWx1YXRpb24gYnVkZ2V0IGFuZCByZXJ1biBwcm92ZW5hbmNlIGFyZSBub3Qgc3BlY2lmaWVkLg","id":"launch-xai-grok47-2248","ingestRunId":"launch-xai-grok47","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-21","rawPayloadHash":"2d9201e3d2f6a14d0a636ed6d0c8aa5b7b3da8c25dccbd911c7d7a2364a66def","report":{"category":"coding","comparabilityReason":"The launch table does not establish matching agent harnesses and evaluation settings. Retained as a provider claim; not a new comparable evaluation.","comparable":false,"configuration":"Grok 4.7 XHigh (DeepSWE: High); Grok 4.6 High; GPT-5.6 Sol Max; Claude Fable 5.1 Max. Launch comparison; per-model agent harness, evaluation budget and rerun provenance are not specified.","family":"Terminal-Bench","locator":"Model Improvements table: Terminal-Bench","metric":"%","notes":"Official launch table. Effort follows the column header except the Grok 4.7 DeepSWE asterisk, which explicitly means High.","sourceId":"xai-grok47","sourceTitle":"Grok 4.7 launch comparison"},"score":37.3,"scoreUnit":"percent","sourceUrl":"https://x.ai/news/grok-4-7"},{"benchmarkId":"terminal-bench-4-0-pct","evidenceKind":"lab_self_report","harnessId":"terminal-bench-4-0-pct:xai-grok47:R3JvayA0LjcgWEhpZ2ggKERlZXBTV0U6IEhpZ2gpOyBHcm9rIDQuNiBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IENsYXVkZSBGYWJsZSA1LjEgTWF4LiBMYXVuY2ggY29tcGFyaXNvbjsgcGVyLW1vZGVsIGFnZW50IGhhcm5lc3MsIGV2YWx1YXRpb24gYnVkZ2V0IGFuZCByZXJ1biBwcm92ZW5hbmNlIGFyZSBub3Qgc3BlY2lmaWVkLg","id":"launch-xai-grok47-2249","ingestRunId":"launch-xai-grok47","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-21","rawPayloadHash":"2d9201e3d2f6a14d0a636ed6d0c8aa5b7b3da8c25dccbd911c7d7a2364a66def","report":{"category":"coding","comparabilityReason":"The launch table does not establish matching agent harnesses and evaluation settings. Retained as a provider claim; not a new comparable evaluation.","comparable":false,"configuration":"Grok 4.7 XHigh (DeepSWE: High); Grok 4.6 High; GPT-5.6 Sol Max; Claude Fable 5.1 Max. Launch comparison; per-model agent harness, evaluation budget and rerun provenance are not specified.","family":"Terminal-Bench","locator":"Model Improvements table: Terminal-Bench","metric":"%","notes":"Official launch table. Effort follows the column header except the Grok 4.7 DeepSWE asterisk, which explicitly means High.","sourceId":"xai-grok47","sourceTitle":"Grok 4.7 launch comparison"},"score":57.9,"scoreUnit":"percent","sourceUrl":"https://x.ai/news/grok-4-7"},{"benchmarkId":"harvey-legal-agent-benchmark","evidenceKind":"lab_self_report","harnessId":"harvey-legal-agent-benchmark:xai-grok47:R3JvayA0LjcgWEhpZ2ggKERlZXBTV0U6IEhpZ2gpOyBHcm9rIDQuNiBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IENsYXVkZSBGYWJsZSA1LjEgTWF4LiBMYXVuY2ggY29tcGFyaXNvbjsgcGVyLW1vZGVsIGFnZW50IGhhcm5lc3MsIGV2YWx1YXRpb24gYnVkZ2V0IGFuZCByZXJ1biBwcm92ZW5hbmNlIGFyZSBub3Qgc3BlY2lmaWVkLg","id":"launch-xai-grok47-2250","ingestRunId":"launch-xai-grok47","modelId":"grok-4.7","n":null,"observedOn":"2026-09-21","rawPayloadHash":"2d9201e3d2f6a14d0a636ed6d0c8aa5b7b3da8c25dccbd911c7d7a2364a66def","report":{"category":"agentic","comparabilityReason":"Secondary Harvey leaderboard report; not an additional independent evaluation. Use the original Vals AI evaluation with its reviewed effort settings.","comparable":false,"configuration":"Grok 4.7 XHigh (DeepSWE: High); Grok 4.6 High; GPT-5.6 Sol Max; Claude Fable 5.1 Max. Launch comparison; per-model agent harness, evaluation budget and rerun provenance are not specified.","family":"Harvey Legal Agent Benchmark","locator":"Model Improvements table: Harvey Legal Agent Benchmark","metric":"%","notes":"Official launch table. Effort follows the column header except the Grok 4.7 DeepSWE asterisk, which explicitly means High. Original evaluator: https://www.vals.ai/benchmarks/hlab","sourceId":"xai-grok47","sourceTitle":"Grok 4.7 launch comparison"},"score":19.6,"scoreUnit":"percent","sourceUrl":"https://x.ai/news/grok-4-7"},{"benchmarkId":"harvey-legal-agent-benchmark","evidenceKind":"lab_self_report","harnessId":"harvey-legal-agent-benchmark:xai-grok47:R3JvayA0LjcgWEhpZ2ggKERlZXBTV0U6IEhpZ2gpOyBHcm9rIDQuNiBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IENsYXVkZSBGYWJsZSA1LjEgTWF4LiBMYXVuY2ggY29tcGFyaXNvbjsgcGVyLW1vZGVsIGFnZW50IGhhcm5lc3MsIGV2YWx1YXRpb24gYnVkZ2V0IGFuZCByZXJ1biBwcm92ZW5hbmNlIGFyZSBub3Qgc3BlY2lmaWVkLg","id":"launch-xai-grok47-2251","ingestRunId":"launch-xai-grok47","modelId":"grok-4.6","n":null,"observedOn":"2026-09-21","rawPayloadHash":"2d9201e3d2f6a14d0a636ed6d0c8aa5b7b3da8c25dccbd911c7d7a2364a66def","report":{"category":"agentic","comparabilityReason":"Secondary Harvey leaderboard report; not an additional independent evaluation. Use the original Vals AI evaluation with its reviewed effort settings.","comparable":false,"configuration":"Grok 4.7 XHigh (DeepSWE: High); Grok 4.6 High; GPT-5.6 Sol Max; Claude Fable 5.1 Max. Launch comparison; per-model agent harness, evaluation budget and rerun provenance are not specified.","family":"Harvey Legal Agent Benchmark","locator":"Model Improvements table: Harvey Legal Agent Benchmark","metric":"%","notes":"Official launch table. Effort follows the column header except the Grok 4.7 DeepSWE asterisk, which explicitly means High. Original evaluator: https://www.vals.ai/benchmarks/hlab","sourceId":"xai-grok47","sourceTitle":"Grok 4.7 launch comparison"},"score":15.8,"scoreUnit":"percent","sourceUrl":"https://x.ai/news/grok-4-7"},{"benchmarkId":"harvey-legal-agent-benchmark","evidenceKind":"lab_self_report","harnessId":"harvey-legal-agent-benchmark:xai-grok47:R3JvayA0LjcgWEhpZ2ggKERlZXBTV0U6IEhpZ2gpOyBHcm9rIDQuNiBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IENsYXVkZSBGYWJsZSA1LjEgTWF4LiBMYXVuY2ggY29tcGFyaXNvbjsgcGVyLW1vZGVsIGFnZW50IGhhcm5lc3MsIGV2YWx1YXRpb24gYnVkZ2V0IGFuZCByZXJ1biBwcm92ZW5hbmNlIGFyZSBub3Qgc3BlY2lmaWVkLg","id":"launch-xai-grok47-2252","ingestRunId":"launch-xai-grok47","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-21","rawPayloadHash":"2d9201e3d2f6a14d0a636ed6d0c8aa5b7b3da8c25dccbd911c7d7a2364a66def","report":{"category":"agentic","comparabilityReason":"Secondary Harvey leaderboard report; not an additional independent evaluation. Use the original Vals AI evaluation with its reviewed effort settings.","comparable":false,"configuration":"Grok 4.7 XHigh (DeepSWE: High); Grok 4.6 High; GPT-5.6 Sol Max; Claude Fable 5.1 Max. Launch comparison; per-model agent harness, evaluation budget and rerun provenance are not specified.","family":"Harvey Legal Agent Benchmark","locator":"Model Improvements table: Harvey Legal Agent Benchmark","metric":"%","notes":"Official launch table. Effort follows the column header except the Grok 4.7 DeepSWE asterisk, which explicitly means High. Original evaluator: https://www.vals.ai/benchmarks/hlab","sourceId":"xai-grok47","sourceTitle":"Grok 4.7 launch comparison"},"score":2.5,"scoreUnit":"percent","sourceUrl":"https://x.ai/news/grok-4-7"},{"benchmarkId":"harvey-legal-agent-benchmark","evidenceKind":"lab_self_report","harnessId":"harvey-legal-agent-benchmark:xai-grok47:R3JvayA0LjcgWEhpZ2ggKERlZXBTV0U6IEhpZ2gpOyBHcm9rIDQuNiBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IENsYXVkZSBGYWJsZSA1LjEgTWF4LiBMYXVuY2ggY29tcGFyaXNvbjsgcGVyLW1vZGVsIGFnZW50IGhhcm5lc3MsIGV2YWx1YXRpb24gYnVkZ2V0IGFuZCByZXJ1biBwcm92ZW5hbmNlIGFyZSBub3Qgc3BlY2lmaWVkLg","id":"launch-xai-grok47-2253","ingestRunId":"launch-xai-grok47","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-21","rawPayloadHash":"2d9201e3d2f6a14d0a636ed6d0c8aa5b7b3da8c25dccbd911c7d7a2364a66def","report":{"category":"agentic","comparabilityReason":"Secondary Harvey leaderboard report; not an additional independent evaluation. Use the original Vals AI evaluation with its reviewed effort settings.","comparable":false,"configuration":"Grok 4.7 XHigh (DeepSWE: High); Grok 4.6 High; GPT-5.6 Sol Max; Claude Fable 5.1 Max. Launch comparison; per-model agent harness, evaluation budget and rerun provenance are not specified.","family":"Harvey Legal Agent Benchmark","locator":"Model Improvements table: Harvey Legal Agent Benchmark","metric":"%","notes":"Official launch table. Effort follows the column header except the Grok 4.7 DeepSWE asterisk, which explicitly means High. Original evaluator: https://www.vals.ai/benchmarks/hlab","sourceId":"xai-grok47","sourceTitle":"Grok 4.7 launch comparison"},"score":6.7,"scoreUnit":"percent","sourceUrl":"https://x.ai/news/grok-4-7"},{"benchmarkId":"healthbench-professional","evidenceKind":"lab_self_report","harnessId":"healthbench-professional:xai-grok47:R3JvayA0LjcgWEhpZ2ggKERlZXBTV0U6IEhpZ2gpOyBHcm9rIDQuNiBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IENsYXVkZSBGYWJsZSA1LjEgTWF4LiBMYXVuY2ggY29tcGFyaXNvbjsgcGVyLW1vZGVsIGFnZW50IGhhcm5lc3MsIGV2YWx1YXRpb24gYnVkZ2V0IGFuZCByZXJ1biBwcm92ZW5hbmNlIGFyZSBub3Qgc3BlY2lmaWVkLg","id":"launch-xai-grok47-2254","ingestRunId":"launch-xai-grok47","modelId":"grok-4.7","n":null,"observedOn":"2026-09-21","rawPayloadHash":"2d9201e3d2f6a14d0a636ed6d0c8aa5b7b3da8c25dccbd911c7d7a2364a66def","report":{"category":"knowledge","comparabilityReason":"The launch table does not establish matching agent harnesses and evaluation settings. Retained as a provider claim; not a new comparable evaluation.","comparable":false,"configuration":"Grok 4.7 XHigh (DeepSWE: High); Grok 4.6 High; GPT-5.6 Sol Max; Claude Fable 5.1 Max. Launch comparison; per-model agent harness, evaluation budget and rerun provenance are not specified.","family":"HealthBench Professional","locator":"Model Improvements table: HealthBench Professional","metric":"%","notes":"Official launch table. Effort follows the column header except the Grok 4.7 DeepSWE asterisk, which explicitly means High.","sourceId":"xai-grok47","sourceTitle":"Grok 4.7 launch comparison"},"score":56.7,"scoreUnit":"percent","sourceUrl":"https://x.ai/news/grok-4-7"},{"benchmarkId":"healthbench-professional","evidenceKind":"lab_self_report","harnessId":"healthbench-professional:xai-grok47:R3JvayA0LjcgWEhpZ2ggKERlZXBTV0U6IEhpZ2gpOyBHcm9rIDQuNiBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IENsYXVkZSBGYWJsZSA1LjEgTWF4LiBMYXVuY2ggY29tcGFyaXNvbjsgcGVyLW1vZGVsIGFnZW50IGhhcm5lc3MsIGV2YWx1YXRpb24gYnVkZ2V0IGFuZCByZXJ1biBwcm92ZW5hbmNlIGFyZSBub3Qgc3BlY2lmaWVkLg","id":"launch-xai-grok47-2255","ingestRunId":"launch-xai-grok47","modelId":"grok-4.6","n":null,"observedOn":"2026-09-21","rawPayloadHash":"2d9201e3d2f6a14d0a636ed6d0c8aa5b7b3da8c25dccbd911c7d7a2364a66def","report":{"category":"knowledge","comparabilityReason":"The launch table does not establish matching agent harnesses and evaluation settings. Retained as a provider claim; not a new comparable evaluation.","comparable":false,"configuration":"Grok 4.7 XHigh (DeepSWE: High); Grok 4.6 High; GPT-5.6 Sol Max; Claude Fable 5.1 Max. Launch comparison; per-model agent harness, evaluation budget and rerun provenance are not specified.","family":"HealthBench Professional","locator":"Model Improvements table: HealthBench Professional","metric":"%","notes":"Official launch table. Effort follows the column header except the Grok 4.7 DeepSWE asterisk, which explicitly means High.","sourceId":"xai-grok47","sourceTitle":"Grok 4.7 launch comparison"},"score":48.5,"scoreUnit":"percent","sourceUrl":"https://x.ai/news/grok-4-7"},{"benchmarkId":"healthbench-professional","evidenceKind":"lab_self_report","harnessId":"healthbench-professional:xai-grok47:R3JvayA0LjcgWEhpZ2ggKERlZXBTV0U6IEhpZ2gpOyBHcm9rIDQuNiBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IENsYXVkZSBGYWJsZSA1LjEgTWF4LiBMYXVuY2ggY29tcGFyaXNvbjsgcGVyLW1vZGVsIGFnZW50IGhhcm5lc3MsIGV2YWx1YXRpb24gYnVkZ2V0IGFuZCByZXJ1biBwcm92ZW5hbmNlIGFyZSBub3Qgc3BlY2lmaWVkLg","id":"launch-xai-grok47-2256","ingestRunId":"launch-xai-grok47","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-21","rawPayloadHash":"2d9201e3d2f6a14d0a636ed6d0c8aa5b7b3da8c25dccbd911c7d7a2364a66def","report":{"category":"knowledge","comparabilityReason":"The launch table does not establish matching agent harnesses and evaluation settings. Retained as a provider claim; not a new comparable evaluation.","comparable":false,"configuration":"Grok 4.7 XHigh (DeepSWE: High); Grok 4.6 High; GPT-5.6 Sol Max; Claude Fable 5.1 Max. Launch comparison; per-model agent harness, evaluation budget and rerun provenance are not specified.","family":"HealthBench Professional","locator":"Model Improvements table: HealthBench Professional","metric":"%","notes":"Official launch table. Effort follows the column header except the Grok 4.7 DeepSWE asterisk, which explicitly means High.","sourceId":"xai-grok47","sourceTitle":"Grok 4.7 launch comparison"},"score":60.5,"scoreUnit":"percent","sourceUrl":"https://x.ai/news/grok-4-7"},{"benchmarkId":"healthbench-professional","evidenceKind":"lab_self_report","harnessId":"healthbench-professional:xai-grok47:R3JvayA0LjcgWEhpZ2ggKERlZXBTV0U6IEhpZ2gpOyBHcm9rIDQuNiBIaWdoOyBHUFQtNS42IFNvbCBNYXg7IENsYXVkZSBGYWJsZSA1LjEgTWF4LiBMYXVuY2ggY29tcGFyaXNvbjsgcGVyLW1vZGVsIGFnZW50IGhhcm5lc3MsIGV2YWx1YXRpb24gYnVkZ2V0IGFuZCByZXJ1biBwcm92ZW5hbmNlIGFyZSBub3Qgc3BlY2lmaWVkLg","id":"launch-xai-grok47-2257","ingestRunId":"launch-xai-grok47","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-21","rawPayloadHash":"2d9201e3d2f6a14d0a636ed6d0c8aa5b7b3da8c25dccbd911c7d7a2364a66def","report":{"category":"knowledge","comparabilityReason":"The launch table does not establish matching agent harnesses and evaluation settings. Retained as a provider claim; not a new comparable evaluation.","comparable":false,"configuration":"Grok 4.7 XHigh (DeepSWE: High); Grok 4.6 High; GPT-5.6 Sol Max; Claude Fable 5.1 Max. Launch comparison; per-model agent harness, evaluation budget and rerun provenance are not specified.","family":"HealthBench Professional","locator":"Model Improvements table: HealthBench Professional","metric":"%","notes":"Official launch table. Effort follows the column header except the Grok 4.7 DeepSWE asterisk, which explicitly means High.","sourceId":"xai-grok47","sourceTitle":"Grok 4.7 launch comparison"},"score":62.1,"scoreUnit":"percent","sourceUrl":"https://x.ai/news/grok-4-7"},{"benchmarkId":"gdpval-aa-2","evidenceKind":"lab_self_report","harnessId":"gdpval-aa-2:xai-grok47:R3JvayA0LjcgWEhpZ2g7IEdyb2sgNC42IEhpZ2g7IENsYXVkZSBGYWJsZSA1LjEgTWF4OyBHUFQtNiBBc3RyYSBNYXguIEdEUHZhbCBsYXVuY2ggY2hhcnQu","id":"launch-xai-grok47-2258","ingestRunId":"launch-xai-grok47","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-21","rawPayloadHash":"2d9201e3d2f6a14d0a636ed6d0c8aa5b7b3da8c25dccbd911c7d7a2364a66def","report":{"category":"agentic","comparabilityReason":"Secondary Artificial Analysis leaderboard report; retain the original evaluator once, without treating the launch reprint as additional evidence.","comparable":false,"configuration":"Grok 4.7 XHigh; Grok 4.6 High; Claude Fable 5.1 Max; GPT-6 Astra Max. GDPval launch chart.","family":"GDPval-AA","locator":"Professional knowledge work: GDPval chart","metric":"Elo","notes":"Launch chart rounded Elo. Original evaluator: https://artificialanalysis.ai/evaluations/gdpval-aa","sourceId":"xai-grok47","sourceTitle":"Grok 4.7 launch comparison"},"score":1735,"scoreUnit":"elo","sourceUrl":"https://x.ai/news/grok-4-7"},{"benchmarkId":"gdpval-aa-2","evidenceKind":"lab_self_report","harnessId":"gdpval-aa-2:xai-grok47:R3JvayA0LjcgWEhpZ2g7IEdyb2sgNC42IEhpZ2g7IENsYXVkZSBGYWJsZSA1LjEgTWF4OyBHUFQtNiBBc3RyYSBNYXguIEdEUHZhbCBsYXVuY2ggY2hhcnQu","id":"launch-xai-grok47-2259","ingestRunId":"launch-xai-grok47","modelId":"grok-4.7","n":null,"observedOn":"2026-09-21","rawPayloadHash":"2d9201e3d2f6a14d0a636ed6d0c8aa5b7b3da8c25dccbd911c7d7a2364a66def","report":{"category":"agentic","comparabilityReason":"Secondary Artificial Analysis leaderboard report; retain the original evaluator once, without treating the launch reprint as additional evidence.","comparable":false,"configuration":"Grok 4.7 XHigh; Grok 4.6 High; Claude Fable 5.1 Max; GPT-6 Astra Max. GDPval launch chart.","family":"GDPval-AA","locator":"Professional knowledge work: GDPval chart","metric":"Elo","notes":"Launch chart rounded Elo. Original evaluator: https://artificialanalysis.ai/evaluations/gdpval-aa","sourceId":"xai-grok47","sourceTitle":"Grok 4.7 launch comparison"},"score":1695,"scoreUnit":"elo","sourceUrl":"https://x.ai/news/grok-4-7"},{"benchmarkId":"gdpval-aa-2","evidenceKind":"lab_self_report","harnessId":"gdpval-aa-2:xai-grok47:R3JvayA0LjcgWEhpZ2g7IEdyb2sgNC42IEhpZ2g7IENsYXVkZSBGYWJsZSA1LjEgTWF4OyBHUFQtNiBBc3RyYSBNYXguIEdEUHZhbCBsYXVuY2ggY2hhcnQu","id":"launch-xai-grok47-2260","ingestRunId":"launch-xai-grok47","modelId":"grok-4.6","n":null,"observedOn":"2026-09-21","rawPayloadHash":"2d9201e3d2f6a14d0a636ed6d0c8aa5b7b3da8c25dccbd911c7d7a2364a66def","report":{"category":"agentic","comparabilityReason":"Secondary Artificial Analysis leaderboard report; retain the original evaluator once, without treating the launch reprint as additional evidence.","comparable":false,"configuration":"Grok 4.7 XHigh; Grok 4.6 High; Claude Fable 5.1 Max; GPT-6 Astra Max. GDPval launch chart.","family":"GDPval-AA","locator":"Professional knowledge work: GDPval chart","metric":"Elo","notes":"Launch chart rounded Elo. Original evaluator: https://artificialanalysis.ai/evaluations/gdpval-aa","sourceId":"xai-grok47","sourceTitle":"Grok 4.7 launch comparison"},"score":1605,"scoreUnit":"elo","sourceUrl":"https://x.ai/news/grok-4-7"},{"benchmarkId":"gdpval-aa-2","evidenceKind":"lab_self_report","harnessId":"gdpval-aa-2:xai-grok47:R3JvayA0LjcgWEhpZ2g7IEdyb2sgNC42IEhpZ2g7IENsYXVkZSBGYWJsZSA1LjEgTWF4OyBHUFQtNiBBc3RyYSBNYXguIEdEUHZhbCBsYXVuY2ggY2hhcnQu","id":"launch-xai-grok47-2261","ingestRunId":"launch-xai-grok47","modelId":"gpt-6-astra","n":null,"observedOn":"2026-09-21","rawPayloadHash":"2d9201e3d2f6a14d0a636ed6d0c8aa5b7b3da8c25dccbd911c7d7a2364a66def","report":{"category":"agentic","comparabilityReason":"Secondary Artificial Analysis leaderboard report; retain the original evaluator once, without treating the launch reprint as additional evidence.","comparable":false,"configuration":"Grok 4.7 XHigh; Grok 4.6 High; Claude Fable 5.1 Max; GPT-6 Astra Max. GDPval launch chart.","family":"GDPval-AA","locator":"Professional knowledge work: GDPval chart","metric":"Elo","notes":"Launch chart rounded Elo. Original evaluator: https://artificialanalysis.ai/evaluations/gdpval-aa","sourceId":"xai-grok47","sourceTitle":"Grok 4.7 launch comparison"},"score":1542,"scoreUnit":"elo","sourceUrl":"https://x.ai/news/grok-4-7"},{"benchmarkId":"terminal-bench-2-1","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1198","ingestRunId":"launch-zai","modelId":"glm-5.3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"Terminal-Bench","locator":"Performance table: Terminal Bench 2.1","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":88.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"terminal-bench-2-1","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1199","ingestRunId":"launch-zai","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"Terminal-Bench","locator":"Performance table: Terminal Bench 2.1","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":81,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"terminal-bench-2-1","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1200","ingestRunId":"launch-zai","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"Terminal-Bench","locator":"Performance table: Terminal Bench 2.1","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":88.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"terminal-bench-2-1","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1201","ingestRunId":"launch-zai","modelId":"deepseek-v4-pro","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"Terminal-Bench","locator":"Performance table: Terminal Bench 2.1","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":87.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"terminal-bench-2-1","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1202","ingestRunId":"launch-zai","modelId":"qwen-3.8-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"Terminal-Bench","locator":"Performance table: Terminal Bench 2.1","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":86.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"terminal-bench-2-1","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1203","ingestRunId":"launch-zai","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"Terminal-Bench","locator":"Performance table: Terminal Bench 2.1","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":85,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"terminal-bench-2-1","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1204","ingestRunId":"launch-zai","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","comparabilityReason":"The source column includes fallback execution without a documented exact effort.","comparable":false,"configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"Terminal-Bench","locator":"Performance table: Terminal Bench 2.1","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":88,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"terminal-bench-2-1","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2-1:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1205","ingestRunId":"launch-zai","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"Terminal-Bench","locator":"Performance table: Terminal Bench 2.1","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":88.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"terminal-bench-3-0","evidenceKind":"lab_self_report","harnessId":"terminal-bench-3-0:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1206","ingestRunId":"launch-zai","modelId":"glm-5.3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"Terminal-Bench","locator":"Performance table: Terminal Bench 3.0","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":28.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"terminal-bench-3-0","evidenceKind":"lab_self_report","harnessId":"terminal-bench-3-0:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1207","ingestRunId":"launch-zai","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"Terminal-Bench","locator":"Performance table: Terminal Bench 3.0","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":4.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"terminal-bench-3-0","evidenceKind":"lab_self_report","harnessId":"terminal-bench-3-0:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1208","ingestRunId":"launch-zai","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"Terminal-Bench","locator":"Performance table: Terminal Bench 3.0","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":17.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"terminal-bench-3-0","evidenceKind":"lab_self_report","harnessId":"terminal-bench-3-0:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1209","ingestRunId":"launch-zai","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"Terminal-Bench","locator":"Performance table: Terminal Bench 3.0","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":21.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"terminal-bench-3-0","evidenceKind":"lab_self_report","harnessId":"terminal-bench-3-0:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1210","ingestRunId":"launch-zai","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","comparabilityReason":"The source column includes fallback execution without a documented exact effort.","comparable":false,"configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"Terminal-Bench","locator":"Performance table: Terminal Bench 3.0","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":33.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"terminal-bench-3-0","evidenceKind":"lab_self_report","harnessId":"terminal-bench-3-0:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1211","ingestRunId":"launch-zai","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"Terminal-Bench","locator":"Performance table: Terminal Bench 3.0","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":34.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"deepswe-1-1","evidenceKind":"lab_self_report","harnessId":"deepswe-1-1:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1212","ingestRunId":"launch-zai","modelId":"glm-5.3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"DeepSWE (v1.1)","locator":"Performance table: DeepSWE (v1.1)","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":66.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"deepswe-1-1","evidenceKind":"lab_self_report","harnessId":"deepswe-1-1:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1213","ingestRunId":"launch-zai","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"DeepSWE (v1.1)","locator":"Performance table: DeepSWE (v1.1)","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":46.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"deepswe-1-1","evidenceKind":"lab_self_report","harnessId":"deepswe-1-1:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1214","ingestRunId":"launch-zai","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"DeepSWE (v1.1)","locator":"Performance table: DeepSWE (v1.1)","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":67.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"deepswe-1-1","evidenceKind":"lab_self_report","harnessId":"deepswe-1-1:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1215","ingestRunId":"launch-zai","modelId":"deepseek-v4-pro","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"DeepSWE (v1.1)","locator":"Performance table: DeepSWE (v1.1)","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":62.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"deepswe-1-1","evidenceKind":"lab_self_report","harnessId":"deepswe-1-1:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1216","ingestRunId":"launch-zai","modelId":"qwen-3.8-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"DeepSWE (v1.1)","locator":"Performance table: DeepSWE (v1.1)","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":56.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"deepswe-1-1","evidenceKind":"lab_self_report","harnessId":"deepswe-1-1:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1217","ingestRunId":"launch-zai","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"DeepSWE (v1.1)","locator":"Performance table: DeepSWE (v1.1)","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":58,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"deepswe-1-1","evidenceKind":"lab_self_report","harnessId":"deepswe-1-1:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1218","ingestRunId":"launch-zai","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","comparabilityReason":"The source column includes fallback execution without a documented exact effort.","comparable":false,"configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"DeepSWE (v1.1)","locator":"Performance table: DeepSWE (v1.1)","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":69.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"deepswe-1-1","evidenceKind":"lab_self_report","harnessId":"deepswe-1-1:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1219","ingestRunId":"launch-zai","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"DeepSWE (v1.1)","locator":"Performance table: DeepSWE (v1.1)","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":72.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"nl2repo-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"nl2repo-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1220","ingestRunId":"launch-zai","modelId":"glm-5.3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"NL2Repo","locator":"Performance table: NL2Repo","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":58,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"nl2repo-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"nl2repo-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1221","ingestRunId":"launch-zai","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"NL2Repo","locator":"Performance table: NL2Repo","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":48.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"nl2repo-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"nl2repo-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1222","ingestRunId":"launch-zai","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"NL2Repo","locator":"Performance table: NL2Repo","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":58,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"nl2repo-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"nl2repo-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1223","ingestRunId":"launch-zai","modelId":"deepseek-v4-pro","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"NL2Repo","locator":"Performance table: NL2Repo","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":61.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"nl2repo-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"nl2repo-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1224","ingestRunId":"launch-zai","modelId":"qwen-3.8-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"NL2Repo","locator":"Performance table: NL2Repo","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":55.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"nl2repo-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"nl2repo-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1225","ingestRunId":"launch-zai","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"NL2Repo","locator":"Performance table: NL2Repo","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":69.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"programbench-almost-solved-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"programbench-almost-solved-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1226","ingestRunId":"launch-zai","modelId":"glm-5.3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"ProgramBench (Almost Solved)","locator":"Performance table: ProgramBench (Almost Solved)","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":19,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"programbench-almost-solved-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"programbench-almost-solved-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1227","ingestRunId":"launch-zai","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"ProgramBench (Almost Solved)","locator":"Performance table: ProgramBench (Almost Solved)","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":9.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"programbench-almost-solved-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"programbench-almost-solved-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1228","ingestRunId":"launch-zai","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"ProgramBench (Almost Solved)","locator":"Performance table: ProgramBench (Almost Solved)","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":17.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"programbench-almost-solved-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"programbench-almost-solved-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1229","ingestRunId":"launch-zai","modelId":"qwen-3.8-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"ProgramBench (Almost Solved)","locator":"Performance table: ProgramBench (Almost Solved)","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":10.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"programbench-almost-solved-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"programbench-almost-solved-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1230","ingestRunId":"launch-zai","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"ProgramBench (Almost Solved)","locator":"Performance table: ProgramBench (Almost Solved)","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":15.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"programbench-almost-solved-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"programbench-almost-solved-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1231","ingestRunId":"launch-zai","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","comparabilityReason":"The source column includes fallback execution without a documented exact effort.","comparable":false,"configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"ProgramBench (Almost Solved)","locator":"Performance table: ProgramBench (Almost Solved)","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":33,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"programbench-almost-solved-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"programbench-almost-solved-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1232","ingestRunId":"launch-zai","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"ProgramBench (Almost Solved)","locator":"Performance table: ProgramBench (Almost Solved)","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":23,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"frontierswe-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"frontierswe-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1233","ingestRunId":"launch-zai","modelId":"glm-5.3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","comparabilityReason":"The GLM card attributes this evaluation to Proximal; it is not a new Z.ai comparison.","comparable":false,"configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"FrontierSWE","locator":"Performance table: FrontierSWE","metric":"%","notes":"Externally evaluated result, attributed in the source footnote.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":78.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"frontierswe-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"frontierswe-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1234","ingestRunId":"launch-zai","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","comparabilityReason":"The GLM card attributes this evaluation to Proximal; it is not a new Z.ai comparison.","comparable":false,"configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"FrontierSWE","locator":"Performance table: FrontierSWE","metric":"%","notes":"Externally evaluated result, attributed in the source footnote.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":67.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"frontierswe-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"frontierswe-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1235","ingestRunId":"launch-zai","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","comparabilityReason":"The GLM card attributes this evaluation to Proximal; it is not a new Z.ai comparison.","comparable":false,"configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"FrontierSWE","locator":"Performance table: FrontierSWE","metric":"%","notes":"Externally evaluated result, attributed in the source footnote.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":66.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"frontierswe-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"frontierswe-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1236","ingestRunId":"launch-zai","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","comparabilityReason":"The GLM card attributes this evaluation to Proximal; it is not a new Z.ai comparison. The source column includes fallback execution without a documented exact effort.","comparable":false,"configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"FrontierSWE","locator":"Performance table: FrontierSWE","metric":"%","notes":"Externally evaluated result, attributed in the source footnote.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":88.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"swe-marathon-1-1","evidenceKind":"lab_self_report","harnessId":"swe-marathon-1-1:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1237","ingestRunId":"launch-zai","modelId":"glm-5.3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","comparabilityReason":"The source replaces affected anti-cheat checks with LLM inspection and changes task images; the modified protocol requires separate review.","comparable":false,"configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"SWE-Marathon (v1.1)","locator":"Performance table: SWE-Marathon (v1.1)","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":42.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"swe-marathon-1-1","evidenceKind":"lab_self_report","harnessId":"swe-marathon-1-1:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1238","ingestRunId":"launch-zai","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"SWE-Marathon (v1.1)","locator":"Performance table: SWE-Marathon (v1.1)","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":19.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"swe-marathon-1-1","evidenceKind":"lab_self_report","harnessId":"swe-marathon-1-1:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1239","ingestRunId":"launch-zai","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"SWE-Marathon (v1.1)","locator":"Performance table: SWE-Marathon (v1.1)","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":48.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"swe-marathon-1-1","evidenceKind":"lab_self_report","harnessId":"swe-marathon-1-1:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1240","ingestRunId":"launch-zai","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"SWE-Marathon (v1.1)","locator":"Performance table: SWE-Marathon (v1.1)","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":48.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"swe-marathon-1-1","evidenceKind":"lab_self_report","harnessId":"swe-marathon-1-1:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1241","ingestRunId":"launch-zai","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","comparabilityReason":"The source column includes fallback execution without a documented exact effort.","comparable":false,"configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"SWE-Marathon (v1.1)","locator":"Performance table: SWE-Marathon (v1.1)","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":33.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"swe-marathon-1-1","evidenceKind":"lab_self_report","harnessId":"swe-marathon-1-1:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1242","ingestRunId":"launch-zai","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"SWE-Marathon (v1.1)","locator":"Performance table: SWE-Marathon (v1.1)","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":42.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"posttrainbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"posttrainbench-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1243","ingestRunId":"launch-zai","modelId":"glm-5.3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","comparabilityReason":"The source changes anti-cheat checks and substitutes zero-shot baseline scores for failed runs; the modified protocol requires separate review.","comparable":false,"configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"PostTrainBench","locator":"Performance table: PostTrainBench","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":39.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"posttrainbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"posttrainbench-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1244","ingestRunId":"launch-zai","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"PostTrainBench","locator":"Performance table: PostTrainBench","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":31.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"posttrainbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"posttrainbench-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1245","ingestRunId":"launch-zai","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"PostTrainBench","locator":"Performance table: PostTrainBench","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":32,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"posttrainbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"posttrainbench-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1246","ingestRunId":"launch-zai","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"PostTrainBench","locator":"Performance table: PostTrainBench","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":32.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"posttrainbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"posttrainbench-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1247","ingestRunId":"launch-zai","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","comparabilityReason":"The source column includes fallback execution without a documented exact effort.","comparable":false,"configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"PostTrainBench","locator":"Performance table: PostTrainBench","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":41.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"posttrainbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"posttrainbench-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1248","ingestRunId":"launch-zai","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"PostTrainBench","locator":"Performance table: PostTrainBench","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":36.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"cybergym-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"cybergym-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1249","ingestRunId":"launch-zai","modelId":"glm-5.3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"CyberGym","locator":"Performance table: CyberGym","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":84.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"cybergym-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"cybergym-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1250","ingestRunId":"launch-zai","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"CyberGym","locator":"Performance table: CyberGym","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":77.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"cybergym-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"cybergym-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1251","ingestRunId":"launch-zai","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"CyberGym","locator":"Performance table: CyberGym","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":80,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"cybergym-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"cybergym-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1252","ingestRunId":"launch-zai","modelId":"deepseek-v4-pro","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"CyberGym","locator":"Performance table: CyberGym","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":83.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"cybergym-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"cybergym-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1253","ingestRunId":"launch-zai","modelId":"qwen-3.8-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"CyberGym","locator":"Performance table: CyberGym","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":78.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"cybergym-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"cybergym-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1254","ingestRunId":"launch-zai","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"CyberGym","locator":"Performance table: CyberGym","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":78.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"cybergym-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"cybergym-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1255","ingestRunId":"launch-zai","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","comparabilityReason":"The source column includes fallback execution without a documented exact effort.","comparable":false,"configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"CyberGym","locator":"Performance table: CyberGym","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":83.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"cybergym-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"cybergym-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1256","ingestRunId":"launch-zai","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"CyberGym","locator":"Performance table: CyberGym","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":83.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"exploitgym-2h-6h-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"exploitgym-2h-6h-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4gMmggU2luZ2xlLXJ1bjg2OXRhc2tzOyBBUEkgdGltZSBub3JtYWxpemVkIGJ5IG1vZGVsIFRQUyBwbHVzIG92ZXJoZWFkOyByZXBvcnRlZCBudW1iZXIgc29sdmVkOyBkb21haW4gd2hpdGVsaXN0Lg","id":"launch-zai-1257","ingestRunId":"launch-zai","modelId":"glm-5.3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details. 2h Single-run869tasks; API time normalized by model TPS plus overhead; reported number solved; domain whitelist.","family":"ExploitGym (2h / 6h)","locator":"Performance table: ExploitGym (2h / 6h)","metric":"count","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":105,"scoreUnit":"index","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"exploitgym-2h-6h-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"exploitgym-2h-6h-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4gNmggU2luZ2xlLXJ1bjg2OXRhc2tzOyBBUEkgdGltZSBub3JtYWxpemVkIGJ5IG1vZGVsIFRQUyBwbHVzIG92ZXJoZWFkOyByZXBvcnRlZCBudW1iZXIgc29sdmVkOyBkb21haW4gd2hpdGVsaXN0Lg","id":"launch-zai-1258","ingestRunId":"launch-zai","modelId":"glm-5.3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details. 6h Single-run869tasks; API time normalized by model TPS plus overhead; reported number solved; domain whitelist.","family":"ExploitGym (2h / 6h)","locator":"Performance table: ExploitGym (2h / 6h)","metric":"count","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":130,"scoreUnit":"index","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"exploitgym-2h-6h-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"exploitgym-2h-6h-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4gMmggU2luZ2xlLXJ1bjg2OXRhc2tzOyBBUEkgdGltZSBub3JtYWxpemVkIGJ5IG1vZGVsIFRQUyBwbHVzIG92ZXJoZWFkOyByZXBvcnRlZCBudW1iZXIgc29sdmVkOyBkb21haW4gd2hpdGVsaXN0Lg","id":"launch-zai-1259","ingestRunId":"launch-zai","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details. 2h Single-run869tasks; API time normalized by model TPS plus overhead; reported number solved; domain whitelist.","family":"ExploitGym (2h / 6h)","locator":"Performance table: ExploitGym (2h / 6h)","metric":"count","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":29,"scoreUnit":"index","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"exploitgym-2h-6h-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"exploitgym-2h-6h-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4gNmggU2luZ2xlLXJ1bjg2OXRhc2tzOyBBUEkgdGltZSBub3JtYWxpemVkIGJ5IG1vZGVsIFRQUyBwbHVzIG92ZXJoZWFkOyByZXBvcnRlZCBudW1iZXIgc29sdmVkOyBkb21haW4gd2hpdGVsaXN0Lg","id":"launch-zai-1260","ingestRunId":"launch-zai","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details. 6h Single-run869tasks; API time normalized by model TPS plus overhead; reported number solved; domain whitelist.","family":"ExploitGym (2h / 6h)","locator":"Performance table: ExploitGym (2h / 6h)","metric":"count","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":39,"scoreUnit":"index","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"exploitgym-2h-6h-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"exploitgym-2h-6h-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4gMmggU2luZ2xlLXJ1bjg2OXRhc2tzOyBBUEkgdGltZSBub3JtYWxpemVkIGJ5IG1vZGVsIFRQUyBwbHVzIG92ZXJoZWFkOyByZXBvcnRlZCBudW1iZXIgc29sdmVkOyBkb21haW4gd2hpdGVsaXN0Lg","id":"launch-zai-1261","ingestRunId":"launch-zai","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details. 2h Single-run869tasks; API time normalized by model TPS plus overhead; reported number solved; domain whitelist.","family":"ExploitGym (2h / 6h)","locator":"Performance table: ExploitGym (2h / 6h)","metric":"count","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":36,"scoreUnit":"index","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"exploitgym-2h-6h-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"exploitgym-2h-6h-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4gNmggU2luZ2xlLXJ1bjg2OXRhc2tzOyBBUEkgdGltZSBub3JtYWxpemVkIGJ5IG1vZGVsIFRQUyBwbHVzIG92ZXJoZWFkOyByZXBvcnRlZCBudW1iZXIgc29sdmVkOyBkb21haW4gd2hpdGVsaXN0Lg","id":"launch-zai-1262","ingestRunId":"launch-zai","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details. 6h Single-run869tasks; API time normalized by model TPS plus overhead; reported number solved; domain whitelist.","family":"ExploitGym (2h / 6h)","locator":"Performance table: ExploitGym (2h / 6h)","metric":"count","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":70,"scoreUnit":"index","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"exploitgym-2h-6h-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"exploitgym-2h-6h-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4gMmggU2luZ2xlLXJ1bjg2OXRhc2tzOyBBUEkgdGltZSBub3JtYWxpemVkIGJ5IG1vZGVsIFRQUyBwbHVzIG92ZXJoZWFkOyByZXBvcnRlZCBudW1iZXIgc29sdmVkOyBkb21haW4gd2hpdGVsaXN0Lg","id":"launch-zai-1263","ingestRunId":"launch-zai","modelId":"qwen-3.8-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details. 2h Single-run869tasks; API time normalized by model TPS plus overhead; reported number solved; domain whitelist.","family":"ExploitGym (2h / 6h)","locator":"Performance table: ExploitGym (2h / 6h)","metric":"count","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":14,"scoreUnit":"index","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"exploitgym-2h-6h-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"exploitgym-2h-6h-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4gNmggU2luZ2xlLXJ1bjg2OXRhc2tzOyBBUEkgdGltZSBub3JtYWxpemVkIGJ5IG1vZGVsIFRQUyBwbHVzIG92ZXJoZWFkOyByZXBvcnRlZCBudW1iZXIgc29sdmVkOyBkb21haW4gd2hpdGVsaXN0Lg","id":"launch-zai-1264","ingestRunId":"launch-zai","modelId":"qwen-3.8-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details. 6h Single-run869tasks; API time normalized by model TPS plus overhead; reported number solved; domain whitelist.","family":"ExploitGym (2h / 6h)","locator":"Performance table: ExploitGym (2h / 6h)","metric":"count","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":26,"scoreUnit":"index","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"exploitgym-2h-6h-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"exploitgym-2h-6h-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4gMmggU2luZ2xlLXJ1bjg2OXRhc2tzOyBBUEkgdGltZSBub3JtYWxpemVkIGJ5IG1vZGVsIFRQUyBwbHVzIG92ZXJoZWFkOyByZXBvcnRlZCBudW1iZXIgc29sdmVkOyBkb21haW4gd2hpdGVsaXN0Lg","id":"launch-zai-1265","ingestRunId":"launch-zai","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details. 2h Single-run869tasks; API time normalized by model TPS plus overhead; reported number solved; domain whitelist.","family":"ExploitGym (2h / 6h)","locator":"Performance table: ExploitGym (2h / 6h)","metric":"count","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":80,"scoreUnit":"index","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"exploitgym-2h-6h-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"exploitgym-2h-6h-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4gNmggU2luZ2xlLXJ1bjg2OXRhc2tzOyBBUEkgdGltZSBub3JtYWxpemVkIGJ5IG1vZGVsIFRQUyBwbHVzIG92ZXJoZWFkOyByZXBvcnRlZCBudW1iZXIgc29sdmVkOyBkb21haW4gd2hpdGVsaXN0Lg","id":"launch-zai-1266","ingestRunId":"launch-zai","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details. 6h Single-run869tasks; API time normalized by model TPS plus overhead; reported number solved; domain whitelist.","family":"ExploitGym (2h / 6h)","locator":"Performance table: ExploitGym (2h / 6h)","metric":"count","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":120,"scoreUnit":"index","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"exploitgym-2h-6h-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"exploitgym-2h-6h-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4gMmggU2luZ2xlLXJ1bjg2OXRhc2tzOyBBUEkgdGltZSBub3JtYWxpemVkIGJ5IG1vZGVsIFRQUyBwbHVzIG92ZXJoZWFkOyByZXBvcnRlZCBudW1iZXIgc29sdmVkOyBkb21haW4gd2hpdGVsaXN0Lg","id":"launch-zai-1267","ingestRunId":"launch-zai","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","comparabilityReason":"The source column includes fallback execution without a documented exact effort.","comparable":false,"configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details. 2h Single-run869tasks; API time normalized by model TPS plus overhead; reported number solved; domain whitelist.","family":"ExploitGym (2h / 6h)","locator":"Performance table: ExploitGym (2h / 6h)","metric":"count","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":181,"scoreUnit":"index","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"exploitgym-2h-6h-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"exploitgym-2h-6h-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4gNmggU2luZ2xlLXJ1bjg2OXRhc2tzOyBBUEkgdGltZSBub3JtYWxpemVkIGJ5IG1vZGVsIFRQUyBwbHVzIG92ZXJoZWFkOyByZXBvcnRlZCBudW1iZXIgc29sdmVkOyBkb21haW4gd2hpdGVsaXN0Lg","id":"launch-zai-1268","ingestRunId":"launch-zai","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","comparabilityReason":"The source column includes fallback execution without a documented exact effort.","comparable":false,"configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details. 6h Single-run869tasks; API time normalized by model TPS plus overhead; reported number solved; domain whitelist.","family":"ExploitGym (2h / 6h)","locator":"Performance table: ExploitGym (2h / 6h)","metric":"count","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":247,"scoreUnit":"index","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"exploitgym-2h-6h-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"exploitgym-2h-6h-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4gMmggU2luZ2xlLXJ1bjg2OXRhc2tzOyBBUEkgdGltZSBub3JtYWxpemVkIGJ5IG1vZGVsIFRQUyBwbHVzIG92ZXJoZWFkOyByZXBvcnRlZCBudW1iZXIgc29sdmVkOyBkb21haW4gd2hpdGVsaXN0Lg","id":"launch-zai-1269","ingestRunId":"launch-zai","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details. 2h Single-run869tasks; API time normalized by model TPS plus overhead; reported number solved; domain whitelist.","family":"ExploitGym (2h / 6h)","locator":"Performance table: ExploitGym (2h / 6h)","metric":"count","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":216,"scoreUnit":"index","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"exploitgym-2h-6h-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"exploitgym-2h-6h-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4gNmggU2luZ2xlLXJ1bjg2OXRhc2tzOyBBUEkgdGltZSBub3JtYWxpemVkIGJ5IG1vZGVsIFRQUyBwbHVzIG92ZXJoZWFkOyByZXBvcnRlZCBudW1iZXIgc29sdmVkOyBkb21haW4gd2hpdGVsaXN0Lg","id":"launch-zai-1270","ingestRunId":"launch-zai","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details. 6h Single-run869tasks; API time normalized by model TPS plus overhead; reported number solved; domain whitelist.","family":"ExploitGym (2h / 6h)","locator":"Performance table: ExploitGym (2h / 6h)","metric":"count","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":293,"scoreUnit":"index","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"exploitbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"exploitbench-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1271","ingestRunId":"launch-zai","modelId":"glm-5.3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"ExploitBench","locator":"Performance table: ExploitBench","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":54.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"exploitbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"exploitbench-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1272","ingestRunId":"launch-zai","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"ExploitBench","locator":"Performance table: ExploitBench","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":24.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"exploitbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"exploitbench-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1273","ingestRunId":"launch-zai","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"ExploitBench","locator":"Performance table: ExploitBench","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":32.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"exploitbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"exploitbench-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1274","ingestRunId":"launch-zai","modelId":"qwen-3.8-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"ExploitBench","locator":"Performance table: ExploitBench","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":28.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"exploitbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"exploitbench-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1275","ingestRunId":"launch-zai","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"ExploitBench","locator":"Performance table: ExploitBench","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":40,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"exploitbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"exploitbench-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1276","ingestRunId":"launch-zai","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","comparabilityReason":"The source column includes fallback execution without a documented exact effort.","comparable":false,"configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"ExploitBench","locator":"Performance table: ExploitBench","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":78,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"exploitbench-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"exploitbench-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1277","ingestRunId":"launch-zai","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"coding","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"ExploitBench","locator":"Performance table: ExploitBench","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":76.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"toolathlon-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"toolathlon-verified-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1278","ingestRunId":"launch-zai","modelId":"glm-5.3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"agentic","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"Toolathlon Verified","locator":"Performance table: Toolathlon Verified","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":73,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"toolathlon-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"toolathlon-verified-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1279","ingestRunId":"launch-zai","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"agentic","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"Toolathlon Verified","locator":"Performance table: Toolathlon Verified","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":59.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"toolathlon-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"toolathlon-verified-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1280","ingestRunId":"launch-zai","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"agentic","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"Toolathlon Verified","locator":"Performance table: Toolathlon Verified","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":76.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"toolathlon-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"toolathlon-verified-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1281","ingestRunId":"launch-zai","modelId":"deepseek-v4-pro","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"agentic","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"Toolathlon Verified","locator":"Performance table: Toolathlon Verified","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":74.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"toolathlon-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"toolathlon-verified-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1282","ingestRunId":"launch-zai","modelId":"qwen-3.8-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"agentic","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"Toolathlon Verified","locator":"Performance table: Toolathlon Verified","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":72.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"toolathlon-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"toolathlon-verified-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1283","ingestRunId":"launch-zai","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"agentic","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"Toolathlon Verified","locator":"Performance table: Toolathlon Verified","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":76.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"toolathlon-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"toolathlon-verified-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1284","ingestRunId":"launch-zai","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"agentic","comparabilityReason":"The source column includes fallback execution without a documented exact effort.","comparable":false,"configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"Toolathlon Verified","locator":"Performance table: Toolathlon Verified","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":74.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"toolathlon-verified-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"toolathlon-verified-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1285","ingestRunId":"launch-zai","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"agentic","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"Toolathlon Verified","locator":"Performance table: Toolathlon Verified","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":74.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"automationbench-1-0-6","evidenceKind":"lab_self_report","harnessId":"automationbench-1-0-6:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1286","ingestRunId":"launch-zai","modelId":"glm-5.3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"agentic","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"AutomationBench (v1.0.6)","locator":"Performance table: AutomationBench (v1.0.6)","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":48.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"automationbench-1-0-6","evidenceKind":"lab_self_report","harnessId":"automationbench-1-0-6:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1287","ingestRunId":"launch-zai","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"agentic","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"AutomationBench (v1.0.6)","locator":"Performance table: AutomationBench (v1.0.6)","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":26.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"automationbench-1-0-6","evidenceKind":"lab_self_report","harnessId":"automationbench-1-0-6:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1288","ingestRunId":"launch-zai","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"agentic","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"AutomationBench (v1.0.6)","locator":"Performance table: AutomationBench (v1.0.6)","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":46.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"automationbench-1-0-6","evidenceKind":"lab_self_report","harnessId":"automationbench-1-0-6:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1289","ingestRunId":"launch-zai","modelId":"deepseek-v4-pro","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"agentic","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"AutomationBench (v1.0.6)","locator":"Performance table: AutomationBench (v1.0.6)","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":43.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"automationbench-1-0-6","evidenceKind":"lab_self_report","harnessId":"automationbench-1-0-6:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1290","ingestRunId":"launch-zai","modelId":"qwen-3.8-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"agentic","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"AutomationBench (v1.0.6)","locator":"Performance table: AutomationBench (v1.0.6)","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":39.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"automationbench-1-0-6","evidenceKind":"lab_self_report","harnessId":"automationbench-1-0-6:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1291","ingestRunId":"launch-zai","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"agentic","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"AutomationBench (v1.0.6)","locator":"Performance table: AutomationBench (v1.0.6)","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":41,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"automationbench-1-0-6","evidenceKind":"lab_self_report","harnessId":"automationbench-1-0-6:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1292","ingestRunId":"launch-zai","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"agentic","comparabilityReason":"The source column includes fallback execution without a documented exact effort.","comparable":false,"configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"AutomationBench (v1.0.6)","locator":"Performance table: AutomationBench (v1.0.6)","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":46.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"automationbench-1-0-6","evidenceKind":"lab_self_report","harnessId":"automationbench-1-0-6:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1293","ingestRunId":"launch-zai","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"agentic","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"AutomationBench (v1.0.6)","locator":"Performance table: AutomationBench (v1.0.6)","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":45.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"agents-last-exam-ale-cli-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"agents-last-exam-ale-cli-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1294","ingestRunId":"launch-zai","modelId":"glm-5.3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"agentic","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"Agents' Last Exam (ALE-CLI)","locator":"Performance table: Agents' Last Exam (ALE-CLI)","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":28.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"agents-last-exam-ale-cli-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"agents-last-exam-ale-cli-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1295","ingestRunId":"launch-zai","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"agentic","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"Agents' Last Exam (ALE-CLI)","locator":"Performance table: Agents' Last Exam (ALE-CLI)","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":23.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"agents-last-exam-ale-cli-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"agents-last-exam-ale-cli-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1296","ingestRunId":"launch-zai","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"agentic","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"Agents' Last Exam (ALE-CLI)","locator":"Performance table: Agents' Last Exam (ALE-CLI)","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":27.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"agents-last-exam-ale-cli-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"agents-last-exam-ale-cli-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1297","ingestRunId":"launch-zai","modelId":"deepseek-v4-pro","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"agentic","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"Agents' Last Exam (ALE-CLI)","locator":"Performance table: Agents' Last Exam (ALE-CLI)","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":25.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"agents-last-exam-ale-cli-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"agents-last-exam-ale-cli-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1298","ingestRunId":"launch-zai","modelId":"qwen-3.8-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"agentic","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"Agents' Last Exam (ALE-CLI)","locator":"Performance table: Agents' Last Exam (ALE-CLI)","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":27,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"agents-last-exam-ale-cli-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"agents-last-exam-ale-cli-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1299","ingestRunId":"launch-zai","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"agentic","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"Agents' Last Exam (ALE-CLI)","locator":"Performance table: Agents' Last Exam (ALE-CLI)","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":25.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"agents-last-exam-ale-cli-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"agents-last-exam-ale-cli-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1300","ingestRunId":"launch-zai","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"agentic","comparabilityReason":"The source column includes fallback execution without a documented exact effort.","comparable":false,"configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"Agents' Last Exam (ALE-CLI)","locator":"Performance table: Agents' Last Exam (ALE-CLI)","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":23.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"agents-last-exam-ale-cli-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"agents-last-exam-ale-cli-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1301","ingestRunId":"launch-zai","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"agentic","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"Agents' Last Exam (ALE-CLI)","locator":"Performance table: Agents' Last Exam (ALE-CLI)","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":28.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"hle-w-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hle-w-tools-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1302","ingestRunId":"launch-zai","modelId":"glm-5.3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"hard_reasoning","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"Humanity's Last Exam","locator":"Performance table: HLE w/ Tools","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":62.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"hle-w-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hle-w-tools-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1303","ingestRunId":"launch-zai","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"hard_reasoning","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"Humanity's Last Exam","locator":"Performance table: HLE w/ Tools","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":54.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"hle-w-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hle-w-tools-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1304","ingestRunId":"launch-zai","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"hard_reasoning","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"Humanity's Last Exam","locator":"Performance table: HLE w/ Tools","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":59.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"hle-w-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hle-w-tools-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1305","ingestRunId":"launch-zai","modelId":"deepseek-v4-pro","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"hard_reasoning","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"Humanity's Last Exam","locator":"Performance table: HLE w/ Tools","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":60,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"hle-w-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hle-w-tools-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1306","ingestRunId":"launch-zai","modelId":"qwen-3.8-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"hard_reasoning","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"Humanity's Last Exam","locator":"Performance table: HLE w/ Tools","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":56.2,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"hle-w-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hle-w-tools-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1307","ingestRunId":"launch-zai","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"hard_reasoning","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"Humanity's Last Exam","locator":"Performance table: HLE w/ Tools","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":57.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"hle-w-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hle-w-tools-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1308","ingestRunId":"launch-zai","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"hard_reasoning","comparabilityReason":"The source column includes fallback execution without a documented exact effort.","comparable":false,"configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"Humanity's Last Exam","locator":"Performance table: HLE w/ Tools","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":63.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"hle-w-tools-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"hle-w-tools-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1309","ingestRunId":"launch-zai","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"hard_reasoning","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"Humanity's Last Exam","locator":"Performance table: HLE w/ Tools","metric":"%","notes":"First-party reported result.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":64.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"gdpval-aa-v2-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gdpval-aa-v2-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1310","ingestRunId":"launch-zai","modelId":"glm-5.3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"agentic","comparabilityReason":"The GLM card attributes this evaluation to Artificial Analysis; it is not a new Z.ai comparison.","comparable":false,"configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"GDPval-AA v2","locator":"Performance table: GDPval-AA v2","metric":"Elo","notes":"Externally evaluated result, attributed in the source footnote.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":1769,"scoreUnit":"elo","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"gdpval-aa-v2-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gdpval-aa-v2-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1311","ingestRunId":"launch-zai","modelId":"glm-5.2","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"agentic","comparabilityReason":"The GLM card attributes this evaluation to Artificial Analysis; it is not a new Z.ai comparison.","comparable":false,"configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"GDPval-AA v2","locator":"Performance table: GDPval-AA v2","metric":"Elo","notes":"Externally evaluated result, attributed in the source footnote.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":1508,"scoreUnit":"elo","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"gdpval-aa-v2-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gdpval-aa-v2-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1312","ingestRunId":"launch-zai","modelId":"kimi-k3","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"agentic","comparabilityReason":"The GLM card attributes this evaluation to Artificial Analysis; it is not a new Z.ai comparison.","comparable":false,"configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"GDPval-AA v2","locator":"Performance table: GDPval-AA v2","metric":"Elo","notes":"Externally evaluated result, attributed in the source footnote.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":1682,"scoreUnit":"elo","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"gdpval-aa-v2-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gdpval-aa-v2-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1313","ingestRunId":"launch-zai","modelId":"deepseek-v4-pro","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"agentic","comparabilityReason":"The GLM card attributes this evaluation to Artificial Analysis; it is not a new Z.ai comparison.","comparable":false,"configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"GDPval-AA v2","locator":"Performance table: GDPval-AA v2","metric":"Elo","notes":"Externally evaluated result, attributed in the source footnote.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":1590,"scoreUnit":"elo","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"gdpval-aa-v2-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gdpval-aa-v2-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1314","ingestRunId":"launch-zai","modelId":"qwen-3.8-max","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"agentic","comparabilityReason":"The GLM card attributes this evaluation to Artificial Analysis; it is not a new Z.ai comparison.","comparable":false,"configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"GDPval-AA v2","locator":"Performance table: GDPval-AA v2","metric":"Elo","notes":"Externally evaluated result, attributed in the source footnote.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":1739,"scoreUnit":"elo","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"gdpval-aa-v2-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gdpval-aa-v2-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1315","ingestRunId":"launch-zai","modelId":"claude-opus-4-8","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"agentic","comparabilityReason":"The GLM card attributes this evaluation to Artificial Analysis; it is not a new Z.ai comparison.","comparable":false,"configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"GDPval-AA v2","locator":"Performance table: GDPval-AA v2","metric":"Elo","notes":"Externally evaluated result, attributed in the source footnote.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":1588,"scoreUnit":"elo","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"gdpval-aa-v2-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gdpval-aa-v2-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1316","ingestRunId":"launch-zai","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"agentic","comparabilityReason":"The GLM card attributes this evaluation to Artificial Analysis; it is not a new Z.ai comparison. The source column includes fallback execution without a documented exact effort.","comparable":false,"configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"GDPval-AA v2","locator":"Performance table: GDPval-AA v2","metric":"Elo","notes":"Externally evaluated result, attributed in the source footnote.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":1743,"scoreUnit":"elo","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"gdpval-aa-v2-source-release-snapshot-version-not-specified","evidenceKind":"lab_self_report","harnessId":"gdpval-aa-v2-source-release-snapshot-version-not-specified:zai:R0xNLTUuMyByZWxlYXNlIHRhYmxlOyBiZW5jaG1hcmstc3BlY2lmaWMgaGFybmVzcyBhbmQgcmVhc29uaW5nIGNvbmZpZ3VyYXRpb25zIGluIEV2YWx1YXRpb24gRGV0YWlscy4","id":"launch-zai-1317","ingestRunId":"launch-zai","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-09-30","rawPayloadHash":"ed1c0a4563c437a32f8953d637a1db9bd831de1b24bd3062a5ace9e04340b1cf","report":{"category":"agentic","comparabilityReason":"The GLM card attributes this evaluation to Artificial Analysis; it is not a new Z.ai comparison.","comparable":false,"configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details.","family":"GDPval-AA v2","locator":"Performance table: GDPval-AA v2","metric":"Elo","notes":"Externally evaluated result, attributed in the source footnote.","sourceId":"zai","sourceTitle":"zai-org/GLM-5.3"},"score":1730,"scoreUnit":"elo","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md"},{"benchmarkId":"deepswe-v1.1","evidenceKind":"lab_self_report","harnessId":"deepswe-v1.1-anthropic-reported","id":"reported-20260901-claude-fable-5-1-deepswe-v1.1-deepswe-v1.1-anthropic-reported","ingestRunId":"ingest-reported-2026-09-06","modelId":"claude-fable-5-1","n":113,"observedOn":"2026-09-01","rawPayloadHash":null,"score":67.4,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"hle","evidenceKind":"lab_self_report","harnessId":"hle-no-tools","id":"reported-20260901-claude-fable-5-1-hle-hle-no-tools","ingestRunId":"ingest-reported-2026-09-06","modelId":"claude-fable-5-1","n":2500,"observedOn":"2026-09-01","rawPayloadHash":null,"score":60.9,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"hle","evidenceKind":"lab_self_report","harnessId":"hle-with-tools","id":"reported-20260901-claude-fable-5-1-hle-hle-with-tools","ingestRunId":"ingest-reported-2026-09-06","modelId":"claude-fable-5-1","n":2500,"observedOn":"2026-09-01","rawPayloadHash":null,"score":65,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"swe-bench-pro","evidenceKind":"lab_self_report","harnessId":"swe-bench-pro-anthropic-reported","id":"reported-20260901-claude-fable-5-1-swe-bench-pro-swe-bench-pro-anthropic-reported","ingestRunId":"ingest-reported-2026-09-06","modelId":"claude-fable-5-1","n":null,"observedOn":"2026-09-01","rawPayloadHash":null,"score":81.2,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"hle","evidenceKind":"lab_self_report","harnessId":"hle-no-tools","id":"reported-20260901-claude-fable-5-hle-hle-no-tools","ingestRunId":"ingest-reported-2026-09-06","modelId":"claude-fable-5","n":2500,"observedOn":"2026-09-01","rawPayloadHash":null,"score":57.8,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"hle","evidenceKind":"lab_self_report","harnessId":"hle-with-tools","id":"reported-20260901-claude-fable-5-hle-hle-with-tools","ingestRunId":"ingest-reported-2026-09-06","modelId":"claude-fable-5","n":2500,"observedOn":"2026-09-01","rawPayloadHash":null,"score":63.8,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"swe-bench-pro","evidenceKind":"lab_self_report","harnessId":"swe-bench-pro-anthropic-reported","id":"reported-20260901-claude-fable-5-swe-bench-pro-swe-bench-pro-anthropic-reported","ingestRunId":"ingest-reported-2026-09-06","modelId":"claude-fable-5","n":null,"observedOn":"2026-09-01","rawPayloadHash":null,"score":80,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"hle","evidenceKind":"lab_self_report","harnessId":"hle-no-tools","id":"reported-20260901-claude-opus-5-hle-hle-no-tools","ingestRunId":"ingest-reported-2026-09-06","modelId":"claude-opus-5","n":2500,"observedOn":"2026-09-01","rawPayloadHash":null,"score":56.6,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"hle","evidenceKind":"lab_self_report","harnessId":"hle-with-tools","id":"reported-20260901-claude-opus-5-hle-hle-with-tools","ingestRunId":"ingest-reported-2026-09-06","modelId":"claude-opus-5","n":2500,"observedOn":"2026-09-01","rawPayloadHash":null,"score":63.6,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card"},{"benchmarkId":"arc-agi-2","evidenceKind":"official_board","harnessId":"arc-agi-2-official","id":"s-arc2-claude-fable-5-1-ad8fb82d3c8d","ingestRunId":"ingest-arc-agi-2-ad8fb82d3c8d","modelId":"claude-fable-5-1","n":120,"observedOn":"2026-09-01","rawPayloadHash":"ad8fb82d3c8d3c6bf0d6db10613c7b9fb882a5278c1fbf5833ee140d01c5e38e","score":90,"scoreUnit":"percent","sourceUrl":"https://arcprize.org/leaderboard"},{"benchmarkId":"arc-agi-2","evidenceKind":"official_board","harnessId":"arc-agi-2-official","id":"s-arc2-claude-fable-5-ad8fb82d3c8d","ingestRunId":"ingest-arc-agi-2-ad8fb82d3c8d","modelId":"claude-fable-5","n":120,"observedOn":"2026-06-09","rawPayloadHash":"ad8fb82d3c8d3c6bf0d6db10613c7b9fb882a5278c1fbf5833ee140d01c5e38e","score":89.2,"scoreUnit":"percent","sourceUrl":"https://arcprize.org/leaderboard"},{"benchmarkId":"arc-agi-2","evidenceKind":"official_board","harnessId":"arc-agi-2-official","id":"s-arc2-claude-haiku-4-5-20251001-ad8fb82d3c8d","ingestRunId":"ingest-arc-agi-2-ad8fb82d3c8d","modelId":"claude-haiku-4-5-20251001","n":120,"observedOn":"2025-10-01","rawPayloadHash":"ad8fb82d3c8d3c6bf0d6db10613c7b9fb882a5278c1fbf5833ee140d01c5e38e","score":1.3,"scoreUnit":"percent","sourceUrl":"https://arcprize.org/leaderboard"},{"benchmarkId":"arc-agi-2","evidenceKind":"official_board","harnessId":"arc-agi-2-official","id":"s-arc2-claude-opus-4-8-ad8fb82d3c8d","ingestRunId":"ingest-arc-agi-2-ad8fb82d3c8d","modelId":"claude-opus-4-8","n":120,"observedOn":"2026-06-01","rawPayloadHash":"ad8fb82d3c8d3c6bf0d6db10613c7b9fb882a5278c1fbf5833ee140d01c5e38e","score":72.1,"scoreUnit":"percent","sourceUrl":"https://arcprize.org/leaderboard"},{"benchmarkId":"arc-agi-2","evidenceKind":"official_board","harnessId":"arc-agi-2-official","id":"s-arc2-claude-opus-5-5-ad8fb82d3c8d","ingestRunId":"ingest-arc-agi-2-ad8fb82d3c8d","modelId":"claude-opus-5-5","n":120,"observedOn":"2026-09-22","rawPayloadHash":"ad8fb82d3c8d3c6bf0d6db10613c7b9fb882a5278c1fbf5833ee140d01c5e38e","score":93.3,"scoreUnit":"percent","sourceUrl":"https://arcprize.org/leaderboard"},{"benchmarkId":"arc-agi-2","evidenceKind":"official_board","harnessId":"arc-agi-2-official","id":"s-arc2-claude-opus-5-ad8fb82d3c8d","ingestRunId":"ingest-arc-agi-2-ad8fb82d3c8d","modelId":"claude-opus-5","n":120,"observedOn":"2026-07-24","rawPayloadHash":"ad8fb82d3c8d3c6bf0d6db10613c7b9fb882a5278c1fbf5833ee140d01c5e38e","score":90.4,"scoreUnit":"percent","sourceUrl":"https://arcprize.org/leaderboard"},{"benchmarkId":"arc-agi-2","evidenceKind":"official_board","harnessId":"arc-agi-2-official","id":"s-arc2-claude-sonnet-4-5-20250929-ad8fb82d3c8d","ingestRunId":"ingest-arc-agi-2-ad8fb82d3c8d","modelId":"claude-sonnet-4-5-20250929","n":120,"observedOn":"2025-09-29","rawPayloadHash":"ad8fb82d3c8d3c6bf0d6db10613c7b9fb882a5278c1fbf5833ee140d01c5e38e","score":3.8,"scoreUnit":"percent","sourceUrl":"https://arcprize.org/leaderboard"},{"benchmarkId":"arc-agi-2","evidenceKind":"official_board","harnessId":"arc-agi-2-official","id":"s-arc2-claude-sonnet-4-6-ad8fb82d3c8d","ingestRunId":"ingest-arc-agi-2-ad8fb82d3c8d","modelId":"claude-sonnet-4-6","n":120,"observedOn":"2026-02-17","rawPayloadHash":"ad8fb82d3c8d3c6bf0d6db10613c7b9fb882a5278c1fbf5833ee140d01c5e38e","score":60.4,"scoreUnit":"percent","sourceUrl":"https://arcprize.org/leaderboard"},{"benchmarkId":"arc-agi-2","evidenceKind":"official_board","harnessId":"arc-agi-2-official","id":"s-arc2-deepseek-v3.2-ad8fb82d3c8d","ingestRunId":"ingest-arc-agi-2-ad8fb82d3c8d","modelId":"deepseek-v3.2","n":120,"observedOn":"2025-12-01","rawPayloadHash":"ad8fb82d3c8d3c6bf0d6db10613c7b9fb882a5278c1fbf5833ee140d01c5e38e","score":4,"scoreUnit":"percent","sourceUrl":"https://arcprize.org/leaderboard"},{"benchmarkId":"arc-agi-2","evidenceKind":"official_board","harnessId":"arc-agi-2-official","id":"s-arc2-deepseek-v4-pro-ad8fb82d3c8d","ingestRunId":"ingest-arc-agi-2-ad8fb82d3c8d","modelId":"deepseek-v4-pro","n":120,"observedOn":"2026-08-13","rawPayloadHash":"ad8fb82d3c8d3c6bf0d6db10613c7b9fb882a5278c1fbf5833ee140d01c5e38e","score":61.3,"scoreUnit":"percent","sourceUrl":"https://arcprize.org/leaderboard"},{"benchmarkId":"arc-agi-2","evidenceKind":"official_board","harnessId":"arc-agi-2-official","id":"s-arc2-gemini-2.5-flash-ad8fb82d3c8d","ingestRunId":"ingest-arc-agi-2-ad8fb82d3c8d","modelId":"gemini-2.5-flash","n":120,"observedOn":"2025-05-20","rawPayloadHash":"ad8fb82d3c8d3c6bf0d6db10613c7b9fb882a5278c1fbf5833ee140d01c5e38e","score":1.7,"scoreUnit":"percent","sourceUrl":"https://arcprize.org/leaderboard"},{"benchmarkId":"arc-agi-2","evidenceKind":"official_board","harnessId":"arc-agi-2-official","id":"s-arc2-gemini-3-flash-preview-ad8fb82d3c8d","ingestRunId":"ingest-arc-agi-2-ad8fb82d3c8d","modelId":"gemini-3-flash-preview","n":120,"observedOn":"2025-12-17","rawPayloadHash":"ad8fb82d3c8d3c6bf0d6db10613c7b9fb882a5278c1fbf5833ee140d01c5e38e","score":33.6,"scoreUnit":"percent","sourceUrl":"https://arcprize.org/leaderboard"},{"benchmarkId":"arc-agi-2","evidenceKind":"official_board","harnessId":"arc-agi-2-official","id":"s-arc2-gemini-3.1-pro-ad8fb82d3c8d","ingestRunId":"ingest-arc-agi-2-ad8fb82d3c8d","modelId":"gemini-3.1-pro","n":120,"observedOn":"2026-02-19","rawPayloadHash":"ad8fb82d3c8d3c6bf0d6db10613c7b9fb882a5278c1fbf5833ee140d01c5e38e","score":77.1,"scoreUnit":"percent","sourceUrl":"https://arcprize.org/leaderboard"},{"benchmarkId":"arc-agi-2","evidenceKind":"official_board","harnessId":"arc-agi-2-official","id":"s-arc2-gemini-3.5-flash-ad8fb82d3c8d","ingestRunId":"ingest-arc-agi-2-ad8fb82d3c8d","modelId":"gemini-3.5-flash","n":120,"observedOn":"2026-05-19","rawPayloadHash":"ad8fb82d3c8d3c6bf0d6db10613c7b9fb882a5278c1fbf5833ee140d01c5e38e","score":72.1,"scoreUnit":"percent","sourceUrl":"https://arcprize.org/leaderboard"},{"benchmarkId":"arc-agi-2","evidenceKind":"official_board","harnessId":"arc-agi-2-official","id":"s-arc2-gemini-3.5-flash-lite-ad8fb82d3c8d","ingestRunId":"ingest-arc-agi-2-ad8fb82d3c8d","modelId":"gemini-3.5-flash-lite","n":120,"observedOn":"2026-07-21","rawPayloadHash":"ad8fb82d3c8d3c6bf0d6db10613c7b9fb882a5278c1fbf5833ee140d01c5e38e","score":10.3,"scoreUnit":"percent","sourceUrl":"https://arcprize.org/leaderboard"},{"benchmarkId":"arc-agi-2","evidenceKind":"official_board","harnessId":"arc-agi-2-official","id":"s-arc2-gemini-3.6-flash-ad8fb82d3c8d","ingestRunId":"ingest-arc-agi-2-ad8fb82d3c8d","modelId":"gemini-3.6-flash","n":120,"observedOn":"2026-07-21","rawPayloadHash":"ad8fb82d3c8d3c6bf0d6db10613c7b9fb882a5278c1fbf5833ee140d01c5e38e","score":60.4,"scoreUnit":"percent","sourceUrl":"https://arcprize.org/leaderboard"},{"benchmarkId":"arc-agi-2","evidenceKind":"official_board","harnessId":"arc-agi-2-official","id":"s-arc2-gemini-3.7-flash-ad8fb82d3c8d","ingestRunId":"ingest-arc-agi-2-ad8fb82d3c8d","modelId":"gemini-3.7-flash","n":120,"observedOn":"2026-08-13","rawPayloadHash":"ad8fb82d3c8d3c6bf0d6db10613c7b9fb882a5278c1fbf5833ee140d01c5e38e","score":84.6,"scoreUnit":"percent","sourceUrl":"https://arcprize.org/leaderboard"},{"benchmarkId":"arc-agi-2","evidenceKind":"official_board","harnessId":"arc-agi-2-official","id":"s-arc2-gemini-3.8-flash-ad8fb82d3c8d","ingestRunId":"ingest-arc-agi-2-ad8fb82d3c8d","modelId":"gemini-3.8-flash","n":120,"observedOn":"2026-09-02","rawPayloadHash":"ad8fb82d3c8d3c6bf0d6db10613c7b9fb882a5278c1fbf5833ee140d01c5e38e","score":89.2,"scoreUnit":"percent","sourceUrl":"https://arcprize.org/leaderboard"},{"benchmarkId":"arc-agi-2","evidenceKind":"official_board","harnessId":"arc-agi-2-official","id":"s-arc2-glm-5-ad8fb82d3c8d","ingestRunId":"ingest-arc-agi-2-ad8fb82d3c8d","modelId":"glm-5","n":120,"observedOn":"2026-02-20","rawPayloadHash":"ad8fb82d3c8d3c6bf0d6db10613c7b9fb882a5278c1fbf5833ee140d01c5e38e","score":4.9,"scoreUnit":"percent","sourceUrl":"https://arcprize.org/leaderboard"},{"benchmarkId":"arc-agi-2","evidenceKind":"official_board","harnessId":"arc-agi-2-official","id":"s-arc2-glm-5.2-ad8fb82d3c8d","ingestRunId":"ingest-arc-agi-2-ad8fb82d3c8d","modelId":"glm-5.2","n":120,"observedOn":"2026-06-13","rawPayloadHash":"ad8fb82d3c8d3c6bf0d6db10613c7b9fb882a5278c1fbf5833ee140d01c5e38e","score":22.8,"scoreUnit":"percent","sourceUrl":"https://arcprize.org/leaderboard"},{"benchmarkId":"arc-agi-2","evidenceKind":"official_board","harnessId":"arc-agi-2-official","id":"s-arc2-glm-5.3-flash-ad8fb82d3c8d","ingestRunId":"ingest-arc-agi-2-ad8fb82d3c8d","modelId":"glm-5.3-flash","n":120,"observedOn":"2026-08-26","rawPayloadHash":"ad8fb82d3c8d3c6bf0d6db10613c7b9fb882a5278c1fbf5833ee140d01c5e38e","score":65.8,"scoreUnit":"percent","sourceUrl":"https://arcprize.org/leaderboard"},{"benchmarkId":"arc-agi-2","evidenceKind":"official_board","harnessId":"arc-agi-2-official","id":"s-arc2-gpt-4.1-ad8fb82d3c8d","ingestRunId":"ingest-arc-agi-2-ad8fb82d3c8d","modelId":"gpt-4.1","n":120,"observedOn":"2025-04-14","rawPayloadHash":"ad8fb82d3c8d3c6bf0d6db10613c7b9fb882a5278c1fbf5833ee140d01c5e38e","score":0.4,"scoreUnit":"percent","sourceUrl":"https://arcprize.org/leaderboard"},{"benchmarkId":"arc-agi-2","evidenceKind":"official_board","harnessId":"arc-agi-2-official","id":"s-arc2-gpt-4.1-mini-ad8fb82d3c8d","ingestRunId":"ingest-arc-agi-2-ad8fb82d3c8d","modelId":"gpt-4.1-mini","n":120,"observedOn":"2025-04-14","rawPayloadHash":"ad8fb82d3c8d3c6bf0d6db10613c7b9fb882a5278c1fbf5833ee140d01c5e38e","score":0,"scoreUnit":"percent","sourceUrl":"https://arcprize.org/leaderboard"},{"benchmarkId":"arc-agi-2","evidenceKind":"official_board","harnessId":"arc-agi-2-official","id":"s-arc2-gpt-4.1-nano-ad8fb82d3c8d","ingestRunId":"ingest-arc-agi-2-ad8fb82d3c8d","modelId":"gpt-4.1-nano","n":120,"observedOn":"2025-04-14","rawPayloadHash":"ad8fb82d3c8d3c6bf0d6db10613c7b9fb882a5278c1fbf5833ee140d01c5e38e","score":0,"scoreUnit":"percent","sourceUrl":"https://arcprize.org/leaderboard"},{"benchmarkId":"arc-agi-2","evidenceKind":"official_board","harnessId":"arc-agi-2-official","id":"s-arc2-gpt-4o-ad8fb82d3c8d","ingestRunId":"ingest-arc-agi-2-ad8fb82d3c8d","modelId":"gpt-4o","n":120,"observedOn":"2024-11-20","rawPayloadHash":"ad8fb82d3c8d3c6bf0d6db10613c7b9fb882a5278c1fbf5833ee140d01c5e38e","score":0,"scoreUnit":"percent","sourceUrl":"https://arcprize.org/leaderboard"},{"benchmarkId":"arc-agi-2","evidenceKind":"official_board","harnessId":"arc-agi-2-official","id":"s-arc2-gpt-4o-mini-ad8fb82d3c8d","ingestRunId":"ingest-arc-agi-2-ad8fb82d3c8d","modelId":"gpt-4o-mini","n":120,"observedOn":"2024-07-18","rawPayloadHash":"ad8fb82d3c8d3c6bf0d6db10613c7b9fb882a5278c1fbf5833ee140d01c5e38e","score":0,"scoreUnit":"percent","sourceUrl":"https://arcprize.org/leaderboard"},{"benchmarkId":"arc-agi-2","evidenceKind":"official_board","harnessId":"arc-agi-2-official","id":"s-arc2-gpt-5-ad8fb82d3c8d","ingestRunId":"ingest-arc-agi-2-ad8fb82d3c8d","modelId":"gpt-5","n":120,"observedOn":"2025-08-07","rawPayloadHash":"ad8fb82d3c8d3c6bf0d6db10613c7b9fb882a5278c1fbf5833ee140d01c5e38e","score":9.9,"scoreUnit":"percent","sourceUrl":"https://arcprize.org/leaderboard"},{"benchmarkId":"arc-agi-2","evidenceKind":"official_board","harnessId":"arc-agi-2-official","id":"s-arc2-gpt-5-mini-ad8fb82d3c8d","ingestRunId":"ingest-arc-agi-2-ad8fb82d3c8d","modelId":"gpt-5-mini","n":120,"observedOn":"2025-08-07","rawPayloadHash":"ad8fb82d3c8d3c6bf0d6db10613c7b9fb882a5278c1fbf5833ee140d01c5e38e","score":4.4,"scoreUnit":"percent","sourceUrl":"https://arcprize.org/leaderboard"},{"benchmarkId":"arc-agi-2","evidenceKind":"official_board","harnessId":"arc-agi-2-official","id":"s-arc2-gpt-5-nano-ad8fb82d3c8d","ingestRunId":"ingest-arc-agi-2-ad8fb82d3c8d","modelId":"gpt-5-nano","n":120,"observedOn":"2025-08-07","rawPayloadHash":"ad8fb82d3c8d3c6bf0d6db10613c7b9fb882a5278c1fbf5833ee140d01c5e38e","score":2.6,"scoreUnit":"percent","sourceUrl":"https://arcprize.org/leaderboard"},{"benchmarkId":"arc-agi-2","evidenceKind":"official_board","harnessId":"arc-agi-2-official","id":"s-arc2-gpt-5-pro-ad8fb82d3c8d","ingestRunId":"ingest-arc-agi-2-ad8fb82d3c8d","modelId":"gpt-5-pro","n":120,"observedOn":"2025-10-06","rawPayloadHash":"ad8fb82d3c8d3c6bf0d6db10613c7b9fb882a5278c1fbf5833ee140d01c5e38e","score":18.3,"scoreUnit":"percent","sourceUrl":"https://arcprize.org/leaderboard"},{"benchmarkId":"arc-agi-2","evidenceKind":"official_board","harnessId":"arc-agi-2-official","id":"s-arc2-gpt-5.2-ad8fb82d3c8d","ingestRunId":"ingest-arc-agi-2-ad8fb82d3c8d","modelId":"gpt-5.2","n":120,"observedOn":"2025-12-11","rawPayloadHash":"ad8fb82d3c8d3c6bf0d6db10613c7b9fb882a5278c1fbf5833ee140d01c5e38e","score":52.9,"scoreUnit":"percent","sourceUrl":"https://arcprize.org/leaderboard"},{"benchmarkId":"arc-agi-2","evidenceKind":"official_board","harnessId":"arc-agi-2-official","id":"s-arc2-gpt-5.2-pro-ad8fb82d3c8d","ingestRunId":"ingest-arc-agi-2-ad8fb82d3c8d","modelId":"gpt-5.2-pro","n":120,"observedOn":"2025-12-11","rawPayloadHash":"ad8fb82d3c8d3c6bf0d6db10613c7b9fb882a5278c1fbf5833ee140d01c5e38e","score":54.2,"scoreUnit":"percent","sourceUrl":"https://arcprize.org/leaderboard"},{"benchmarkId":"arc-agi-2","evidenceKind":"official_board","harnessId":"arc-agi-2-official","id":"s-arc2-gpt-5.4-ad8fb82d3c8d","ingestRunId":"ingest-arc-agi-2-ad8fb82d3c8d","modelId":"gpt-5.4","n":120,"observedOn":"2026-03-04","rawPayloadHash":"ad8fb82d3c8d3c6bf0d6db10613c7b9fb882a5278c1fbf5833ee140d01c5e38e","score":74,"scoreUnit":"percent","sourceUrl":"https://arcprize.org/leaderboard"},{"benchmarkId":"arc-agi-2","evidenceKind":"official_board","harnessId":"arc-agi-2-official","id":"s-arc2-gpt-5.4-mini-ad8fb82d3c8d","ingestRunId":"ingest-arc-agi-2-ad8fb82d3c8d","modelId":"gpt-5.4-mini","n":120,"observedOn":"2026-03-17","rawPayloadHash":"ad8fb82d3c8d3c6bf0d6db10613c7b9fb882a5278c1fbf5833ee140d01c5e38e","score":18.9,"scoreUnit":"percent","sourceUrl":"https://arcprize.org/leaderboard"},{"benchmarkId":"arc-agi-2","evidenceKind":"official_board","harnessId":"arc-agi-2-official","id":"s-arc2-gpt-5.4-nano-ad8fb82d3c8d","ingestRunId":"ingest-arc-agi-2-ad8fb82d3c8d","modelId":"gpt-5.4-nano","n":120,"observedOn":"2026-03-17","rawPayloadHash":"ad8fb82d3c8d3c6bf0d6db10613c7b9fb882a5278c1fbf5833ee140d01c5e38e","score":5.7,"scoreUnit":"percent","sourceUrl":"https://arcprize.org/leaderboard"},{"benchmarkId":"arc-agi-2","evidenceKind":"official_board","harnessId":"arc-agi-2-official","id":"s-arc2-gpt-5.4-pro-ad8fb82d3c8d","ingestRunId":"ingest-arc-agi-2-ad8fb82d3c8d","modelId":"gpt-5.4-pro","n":120,"observedOn":"2026-03-04","rawPayloadHash":"ad8fb82d3c8d3c6bf0d6db10613c7b9fb882a5278c1fbf5833ee140d01c5e38e","score":83.3,"scoreUnit":"percent","sourceUrl":"https://arcprize.org/leaderboard"},{"benchmarkId":"arc-agi-2","evidenceKind":"official_board","harnessId":"arc-agi-2-official","id":"s-arc2-gpt-5.5-ad8fb82d3c8d","ingestRunId":"ingest-arc-agi-2-ad8fb82d3c8d","modelId":"gpt-5.5","n":120,"observedOn":"2026-04-22","rawPayloadHash":"ad8fb82d3c8d3c6bf0d6db10613c7b9fb882a5278c1fbf5833ee140d01c5e38e","score":85,"scoreUnit":"percent","sourceUrl":"https://arcprize.org/leaderboard"},{"benchmarkId":"arc-agi-2","evidenceKind":"official_board","harnessId":"arc-agi-2-official","id":"s-arc2-gpt-5.5-pro-ad8fb82d3c8d","ingestRunId":"ingest-arc-agi-2-ad8fb82d3c8d","modelId":"gpt-5.5-pro","n":120,"observedOn":"2026-04-23","rawPayloadHash":"ad8fb82d3c8d3c6bf0d6db10613c7b9fb882a5278c1fbf5833ee140d01c5e38e","score":84.6,"scoreUnit":"percent","sourceUrl":"https://arcprize.org/leaderboard"},{"benchmarkId":"arc-agi-2","evidenceKind":"official_board","harnessId":"arc-agi-2-official","id":"s-arc2-gpt-5.6-luna-ad8fb82d3c8d","ingestRunId":"ingest-arc-agi-2-ad8fb82d3c8d","modelId":"gpt-5.6-luna","n":120,"observedOn":"2026-07-09","rawPayloadHash":"ad8fb82d3c8d3c6bf0d6db10613c7b9fb882a5278c1fbf5833ee140d01c5e38e","score":59.5,"scoreUnit":"percent","sourceUrl":"https://arcprize.org/leaderboard"},{"benchmarkId":"arc-agi-2","evidenceKind":"official_board","harnessId":"arc-agi-2-official","id":"s-arc2-gpt-5.6-sol-ad8fb82d3c8d","ingestRunId":"ingest-arc-agi-2-ad8fb82d3c8d","modelId":"gpt-5.6-sol","n":120,"observedOn":"2026-07-09","rawPayloadHash":"ad8fb82d3c8d3c6bf0d6db10613c7b9fb882a5278c1fbf5833ee140d01c5e38e","score":92.5,"scoreUnit":"percent","sourceUrl":"https://arcprize.org/leaderboard"},{"benchmarkId":"arc-agi-2","evidenceKind":"official_board","harnessId":"arc-agi-2-official","id":"s-arc2-gpt-5.6-terra-ad8fb82d3c8d","ingestRunId":"ingest-arc-agi-2-ad8fb82d3c8d","modelId":"gpt-5.6-terra","n":120,"observedOn":"2026-07-09","rawPayloadHash":"ad8fb82d3c8d3c6bf0d6db10613c7b9fb882a5278c1fbf5833ee140d01c5e38e","score":83.9,"scoreUnit":"percent","sourceUrl":"https://arcprize.org/leaderboard"},{"benchmarkId":"arc-agi-2","evidenceKind":"official_board","harnessId":"arc-agi-2-official","id":"s-arc2-gpt-6-astra-ad8fb82d3c8d","ingestRunId":"ingest-arc-agi-2-ad8fb82d3c8d","modelId":"gpt-6-astra","n":120,"observedOn":"2026-09-02","rawPayloadHash":"ad8fb82d3c8d3c6bf0d6db10613c7b9fb882a5278c1fbf5833ee140d01c5e38e","score":95,"scoreUnit":"percent","sourceUrl":"https://arcprize.org/leaderboard"},{"benchmarkId":"arc-agi-2","evidenceKind":"official_board","harnessId":"arc-agi-2-official","id":"s-arc2-gpt-6-luna-ad8fb82d3c8d","ingestRunId":"ingest-arc-agi-2-ad8fb82d3c8d","modelId":"gpt-6-luna","n":120,"observedOn":"2026-09-22","rawPayloadHash":"ad8fb82d3c8d3c6bf0d6db10613c7b9fb882a5278c1fbf5833ee140d01c5e38e","score":59.3,"scoreUnit":"percent","sourceUrl":"https://arcprize.org/leaderboard"},{"benchmarkId":"arc-agi-2","evidenceKind":"official_board","harnessId":"arc-agi-2-official","id":"s-arc2-gpt-6-sol-ad8fb82d3c8d","ingestRunId":"ingest-arc-agi-2-ad8fb82d3c8d","modelId":"gpt-6-sol","n":120,"observedOn":"2026-09-22","rawPayloadHash":"ad8fb82d3c8d3c6bf0d6db10613c7b9fb882a5278c1fbf5833ee140d01c5e38e","score":89.6,"scoreUnit":"percent","sourceUrl":"https://arcprize.org/leaderboard"},{"benchmarkId":"arc-agi-2","evidenceKind":"official_board","harnessId":"arc-agi-2-official","id":"s-arc2-grok-4.6-ad8fb82d3c8d","ingestRunId":"ingest-arc-agi-2-ad8fb82d3c8d","modelId":"grok-4.6","n":120,"observedOn":"2026-08-11","rawPayloadHash":"ad8fb82d3c8d3c6bf0d6db10613c7b9fb882a5278c1fbf5833ee140d01c5e38e","score":67.1,"scoreUnit":"percent","sourceUrl":"https://arcprize.org/leaderboard"},{"benchmarkId":"arc-agi-2","evidenceKind":"official_board","harnessId":"arc-agi-2-official","id":"s-arc2-kimi-k3-ad8fb82d3c8d","ingestRunId":"ingest-arc-agi-2-ad8fb82d3c8d","modelId":"kimi-k3","n":120,"observedOn":"2026-07-16","rawPayloadHash":"ad8fb82d3c8d3c6bf0d6db10613c7b9fb882a5278c1fbf5833ee140d01c5e38e","score":60.4,"scoreUnit":"percent","sourceUrl":"https://arcprize.org/leaderboard"},{"benchmarkId":"arc-agi-2","evidenceKind":"official_board","harnessId":"arc-agi-2-official","id":"s-arc2-llama-4-maverick-ad8fb82d3c8d","ingestRunId":"ingest-arc-agi-2-ad8fb82d3c8d","modelId":"llama-4-maverick","n":120,"observedOn":"2025-04-05","rawPayloadHash":"ad8fb82d3c8d3c6bf0d6db10613c7b9fb882a5278c1fbf5833ee140d01c5e38e","score":0,"scoreUnit":"percent","sourceUrl":"https://arcprize.org/leaderboard"},{"benchmarkId":"arc-agi-2","evidenceKind":"official_board","harnessId":"arc-agi-2-official","id":"s-arc2-minimax-m2.5-ad8fb82d3c8d","ingestRunId":"ingest-arc-agi-2-ad8fb82d3c8d","modelId":"minimax-m2.5","n":120,"observedOn":"2026-02-12","rawPayloadHash":"ad8fb82d3c8d3c6bf0d6db10613c7b9fb882a5278c1fbf5833ee140d01c5e38e","score":4.9,"scoreUnit":"percent","sourceUrl":"https://arcprize.org/leaderboard"},{"benchmarkId":"arc-agi-2","evidenceKind":"official_board","harnessId":"arc-agi-2-official","id":"s-arc2-o3-ad8fb82d3c8d","ingestRunId":"ingest-arc-agi-2-ad8fb82d3c8d","modelId":"o3","n":120,"observedOn":"2025-04-16","rawPayloadHash":"ad8fb82d3c8d3c6bf0d6db10613c7b9fb882a5278c1fbf5833ee140d01c5e38e","score":6.5,"scoreUnit":"percent","sourceUrl":"https://arcprize.org/leaderboard"},{"benchmarkId":"arc-agi-2","evidenceKind":"official_board","harnessId":"arc-agi-2-official","id":"s-arc2-o3-mini-ad8fb82d3c8d","ingestRunId":"ingest-arc-agi-2-ad8fb82d3c8d","modelId":"o3-mini","n":120,"observedOn":"2025-01-31","rawPayloadHash":"ad8fb82d3c8d3c6bf0d6db10613c7b9fb882a5278c1fbf5833ee140d01c5e38e","score":3,"scoreUnit":"percent","sourceUrl":"https://arcprize.org/leaderboard"},{"benchmarkId":"arc-agi-2","evidenceKind":"official_board","harnessId":"arc-agi-2-official","id":"s-arc2-o3-pro-ad8fb82d3c8d","ingestRunId":"ingest-arc-agi-2-ad8fb82d3c8d","modelId":"o3-pro","n":120,"observedOn":"2025-06-10","rawPayloadHash":"ad8fb82d3c8d3c6bf0d6db10613c7b9fb882a5278c1fbf5833ee140d01c5e38e","score":4.9,"scoreUnit":"percent","sourceUrl":"https://arcprize.org/leaderboard"},{"benchmarkId":"arc-agi-2","evidenceKind":"official_board","harnessId":"arc-agi-2-official","id":"s-arc2-o4-mini-ad8fb82d3c8d","ingestRunId":"ingest-arc-agi-2-ad8fb82d3c8d","modelId":"o4-mini","n":120,"observedOn":"2025-04-16","rawPayloadHash":"ad8fb82d3c8d3c6bf0d6db10613c7b9fb882a5278c1fbf5833ee140d01c5e38e","score":6.1,"scoreUnit":"percent","sourceUrl":"https://arcprize.org/leaderboard"},{"benchmarkId":"arc-agi-2","evidenceKind":"official_board","harnessId":"arc-agi-2-official","id":"s-arc2-qwen3-235b-a22b-instruct-2507-ad8fb82d3c8d","ingestRunId":"ingest-arc-agi-2-ad8fb82d3c8d","modelId":"qwen3-235b-a22b-instruct-2507","n":120,"observedOn":"2025-07-25","rawPayloadHash":"ad8fb82d3c8d3c6bf0d6db10613c7b9fb882a5278c1fbf5833ee140d01c5e38e","score":1.3,"scoreUnit":"percent","sourceUrl":"https://arcprize.org/leaderboard"},{"benchmarkId":"automationbench-aa","evidenceKind":"official_board","harnessId":"automationbench-aa-official","id":"s-automationbench-agnes-2.5-pro-alpha-3b0a318dba6f","ingestRunId":"ingest-automationbench-aa-3b0a318dba6f","modelId":"agnes-2.5-pro-alpha","n":657,"observedOn":"2026-09-23","rawPayloadHash":"3b0a318dba6f0de25d6fa36b71a0839c36f512c1da8fd3daef6b199a148a1841","score":38.43295195770323,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/automationbench-aa"},{"benchmarkId":"automationbench-aa","evidenceKind":"official_board","harnessId":"automationbench-aa-official","id":"s-automationbench-claude-fable-5-1-712aa4217411","ingestRunId":"ingest-automationbench-aa-712aa4217411","modelId":"claude-fable-5-1","n":657,"observedOn":"2026-09-29","rawPayloadHash":"712aa421741169313d1deaedce75a8ed04d9eede62542d7613b0e0d3c9e05e6b","score":59.37591715646424,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/automationbench-aa"},{"benchmarkId":"automationbench-aa","evidenceKind":"official_board","harnessId":"automationbench-aa-official","id":"s-automationbench-claude-fable-5-712aa4217411","ingestRunId":"ingest-automationbench-aa-712aa4217411","modelId":"claude-fable-5","n":657,"observedOn":"2026-09-29","rawPayloadHash":"712aa421741169313d1deaedce75a8ed04d9eede62542d7613b0e0d3c9e05e6b","score":54.071301140653816,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/automationbench-aa"},{"benchmarkId":"automationbench-aa","evidenceKind":"official_board","harnessId":"automationbench-aa-official","id":"s-automationbench-claude-opus-4-8-712aa4217411","ingestRunId":"ingest-automationbench-aa-712aa4217411","modelId":"claude-opus-4-8","n":657,"observedOn":"2026-09-29","rawPayloadHash":"712aa421741169313d1deaedce75a8ed04d9eede62542d7613b0e0d3c9e05e6b","score":45.59343691527719,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/automationbench-aa"},{"benchmarkId":"automationbench-aa","evidenceKind":"official_board","harnessId":"automationbench-aa-official","id":"s-automationbench-claude-opus-5-5-712aa4217411","ingestRunId":"ingest-automationbench-aa-712aa4217411","modelId":"claude-opus-5-5","n":657,"observedOn":"2026-09-29","rawPayloadHash":"712aa421741169313d1deaedce75a8ed04d9eede62542d7613b0e0d3c9e05e6b","score":69.5386812863607,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/automationbench-aa"},{"benchmarkId":"automationbench-aa","evidenceKind":"official_board","harnessId":"automationbench-aa-official","id":"s-automationbench-claude-opus-5-712aa4217411","ingestRunId":"ingest-automationbench-aa-712aa4217411","modelId":"claude-opus-5","n":657,"observedOn":"2026-09-29","rawPayloadHash":"712aa421741169313d1deaedce75a8ed04d9eede62542d7613b0e0d3c9e05e6b","score":56.57355576493251,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/automationbench-aa"},{"benchmarkId":"automationbench-aa","evidenceKind":"official_board","harnessId":"automationbench-aa-official","id":"s-automationbench-claude-sonnet-5-5-712aa4217411","ingestRunId":"ingest-automationbench-aa-712aa4217411","modelId":"claude-sonnet-5-5","n":657,"observedOn":"2026-09-29","rawPayloadHash":"712aa421741169313d1deaedce75a8ed04d9eede62542d7613b0e0d3c9e05e6b","score":71.32534614553299,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/automationbench-aa"},{"benchmarkId":"automationbench-aa","evidenceKind":"official_board","harnessId":"automationbench-aa-official","id":"s-automationbench-claude-sonnet-5-712aa4217411","ingestRunId":"ingest-automationbench-aa-712aa4217411","modelId":"claude-sonnet-5","n":657,"observedOn":"2026-09-29","rawPayloadHash":"712aa421741169313d1deaedce75a8ed04d9eede62542d7613b0e0d3c9e05e6b","score":36.51267571703639,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/automationbench-aa"},{"benchmarkId":"automationbench-aa","evidenceKind":"official_board","harnessId":"automationbench-aa-official","id":"s-automationbench-deepseek-v4-flash-712aa4217411","ingestRunId":"ingest-automationbench-aa-712aa4217411","modelId":"deepseek-v4-flash","n":657,"observedOn":"2026-09-29","rawPayloadHash":"712aa421741169313d1deaedce75a8ed04d9eede62542d7613b0e0d3c9e05e6b","score":53.96771855133674,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/automationbench-aa"},{"benchmarkId":"automationbench-aa","evidenceKind":"official_board","harnessId":"automationbench-aa-official","id":"s-automationbench-deepseek-v4-pro-0424-712aa4217411","ingestRunId":"ingest-automationbench-aa-712aa4217411","modelId":"deepseek-v4-pro-0424","n":657,"observedOn":"2026-09-29","rawPayloadHash":"712aa421741169313d1deaedce75a8ed04d9eede62542d7613b0e0d3c9e05e6b","score":56.289685574342705,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/automationbench-aa"},{"benchmarkId":"automationbench-aa","evidenceKind":"official_board","harnessId":"automationbench-aa-official","id":"s-automationbench-deepseek-v4-pro-712aa4217411","ingestRunId":"ingest-automationbench-aa-712aa4217411","modelId":"deepseek-v4-pro","n":657,"observedOn":"2026-09-29","rawPayloadHash":"712aa421741169313d1deaedce75a8ed04d9eede62542d7613b0e0d3c9e05e6b","score":56.711655463440025,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/automationbench-aa"},{"benchmarkId":"automationbench-aa","evidenceKind":"official_board","harnessId":"automationbench-aa-official","id":"s-automationbench-devstral-2512-712aa4217411","ingestRunId":"ingest-automationbench-aa-712aa4217411","modelId":"devstral-2512","n":657,"observedOn":"2026-09-29","rawPayloadHash":"712aa421741169313d1deaedce75a8ed04d9eede62542d7613b0e0d3c9e05e6b","score":3.106312467124685,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/automationbench-aa"},{"benchmarkId":"automationbench-aa","evidenceKind":"official_board","harnessId":"automationbench-aa-official","id":"s-automationbench-gemini-2.5-pro-712aa4217411","ingestRunId":"ingest-automationbench-aa-712aa4217411","modelId":"gemini-2.5-pro","n":657,"observedOn":"2026-09-29","rawPayloadHash":"712aa421741169313d1deaedce75a8ed04d9eede62542d7613b0e0d3c9e05e6b","score":2.174946419340874,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/automationbench-aa"},{"benchmarkId":"automationbench-aa","evidenceKind":"official_board","harnessId":"automationbench-aa-official","id":"s-automationbench-gemini-3.1-flash-lite-712aa4217411","ingestRunId":"ingest-automationbench-aa-712aa4217411","modelId":"gemini-3.1-flash-lite","n":657,"observedOn":"2026-09-29","rawPayloadHash":"712aa421741169313d1deaedce75a8ed04d9eede62542d7613b0e0d3c9e05e6b","score":6.796182033119545,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/automationbench-aa"},{"benchmarkId":"automationbench-aa","evidenceKind":"official_board","harnessId":"automationbench-aa-official","id":"s-automationbench-gemini-3.1-pro-712aa4217411","ingestRunId":"ingest-automationbench-aa-712aa4217411","modelId":"gemini-3.1-pro","n":657,"observedOn":"2026-09-29","rawPayloadHash":"712aa421741169313d1deaedce75a8ed04d9eede62542d7613b0e0d3c9e05e6b","score":35.427718398098364,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/automationbench-aa"},{"benchmarkId":"automationbench-aa","evidenceKind":"official_board","harnessId":"automationbench-aa-official","id":"s-automationbench-gemini-3.5-flash-lite-712aa4217411","ingestRunId":"ingest-automationbench-aa-712aa4217411","modelId":"gemini-3.5-flash-lite","n":657,"observedOn":"2026-09-29","rawPayloadHash":"712aa421741169313d1deaedce75a8ed04d9eede62542d7613b0e0d3c9e05e6b","score":25.011270511840955,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/automationbench-aa"},{"benchmarkId":"automationbench-aa","evidenceKind":"official_board","harnessId":"automationbench-aa-official","id":"s-automationbench-gpt-5-mini-712aa4217411","ingestRunId":"ingest-automationbench-aa-712aa4217411","modelId":"gpt-5-mini","n":657,"observedOn":"2026-09-29","rawPayloadHash":"712aa421741169313d1deaedce75a8ed04d9eede62542d7613b0e0d3c9e05e6b","score":6.487466331916088,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/automationbench-aa"},{"benchmarkId":"automationbench-aa","evidenceKind":"official_board","harnessId":"automationbench-aa-official","id":"s-automationbench-gpt-5.6-sol-712aa4217411","ingestRunId":"ingest-automationbench-aa-712aa4217411","modelId":"gpt-5.6-sol","n":657,"observedOn":"2026-09-29","rawPayloadHash":"712aa421741169313d1deaedce75a8ed04d9eede62542d7613b0e0d3c9e05e6b","score":60.08114996276329,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/automationbench-aa"},{"benchmarkId":"automationbench-aa","evidenceKind":"official_board","harnessId":"automationbench-aa-official","id":"s-automationbench-gpt-6-astra-712aa4217411","ingestRunId":"ingest-automationbench-aa-712aa4217411","modelId":"gpt-6-astra","n":657,"observedOn":"2026-09-29","rawPayloadHash":"712aa421741169313d1deaedce75a8ed04d9eede62542d7613b0e0d3c9e05e6b","score":68.49174806688876,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/automationbench-aa"},{"benchmarkId":"automationbench-aa","evidenceKind":"official_board","harnessId":"automationbench-aa-official","id":"s-automationbench-gpt-6-luna-712aa4217411","ingestRunId":"ingest-automationbench-aa-712aa4217411","modelId":"gpt-6-luna","n":657,"observedOn":"2026-09-29","rawPayloadHash":"712aa421741169313d1deaedce75a8ed04d9eede62542d7613b0e0d3c9e05e6b","score":53.16814034483175,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/automationbench-aa"},{"benchmarkId":"automationbench-aa","evidenceKind":"official_board","harnessId":"automationbench-aa-official","id":"s-automationbench-gpt-6-sol-712aa4217411","ingestRunId":"ingest-automationbench-aa-712aa4217411","modelId":"gpt-6-sol","n":657,"observedOn":"2026-09-29","rawPayloadHash":"712aa421741169313d1deaedce75a8ed04d9eede62542d7613b0e0d3c9e05e6b","score":61.63369295959864,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/automationbench-aa"},{"benchmarkId":"automationbench-aa","evidenceKind":"official_board","harnessId":"automationbench-aa-official","id":"s-automationbench-gpt-oss-120b-712aa4217411","ingestRunId":"ingest-automationbench-aa-712aa4217411","modelId":"gpt-oss-120b","n":657,"observedOn":"2026-09-29","rawPayloadHash":"712aa421741169313d1deaedce75a8ed04d9eede62542d7613b0e0d3c9e05e6b","score":0.19906990691605594,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/automationbench-aa"},{"benchmarkId":"automationbench-aa","evidenceKind":"official_board","harnessId":"automationbench-aa-official","id":"s-automationbench-gpt-oss-20b-712aa4217411","ingestRunId":"ingest-automationbench-aa-712aa4217411","modelId":"gpt-oss-20b","n":657,"observedOn":"2026-09-29","rawPayloadHash":"712aa421741169313d1deaedce75a8ed04d9eede62542d7613b0e0d3c9e05e6b","score":0.19111706206154658,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/automationbench-aa"},{"benchmarkId":"automationbench-aa","evidenceKind":"official_board","harnessId":"automationbench-aa-official","id":"s-automationbench-grok-4.6-712aa4217411","ingestRunId":"ingest-automationbench-aa-712aa4217411","modelId":"grok-4.6","n":657,"observedOn":"2026-09-29","rawPayloadHash":"712aa421741169313d1deaedce75a8ed04d9eede62542d7613b0e0d3c9e05e6b","score":66.67849638121398,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/automationbench-aa"},{"benchmarkId":"automationbench-aa","evidenceKind":"official_board","harnessId":"automationbench-aa-official","id":"s-automationbench-grok-4.7-712aa4217411","ingestRunId":"ingest-automationbench-aa-712aa4217411","modelId":"grok-4.7","n":657,"observedOn":"2026-09-29","rawPayloadHash":"712aa421741169313d1deaedce75a8ed04d9eede62542d7613b0e0d3c9e05e6b","score":65.55835148969457,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/automationbench-aa"},{"benchmarkId":"automationbench-aa","evidenceKind":"official_board","harnessId":"automationbench-aa-official","id":"s-automationbench-hy3-712aa4217411","ingestRunId":"ingest-automationbench-aa-712aa4217411","modelId":"hy3","n":657,"observedOn":"2026-09-29","rawPayloadHash":"712aa421741169313d1deaedce75a8ed04d9eede62542d7613b0e0d3c9e05e6b","score":18.409128371172397,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/automationbench-aa"},{"benchmarkId":"automationbench-aa","evidenceKind":"official_board","harnessId":"automationbench-aa-official","id":"s-automationbench-kimi-k3-712aa4217411","ingestRunId":"ingest-automationbench-aa-712aa4217411","modelId":"kimi-k3","n":657,"observedOn":"2026-09-29","rawPayloadHash":"712aa421741169313d1deaedce75a8ed04d9eede62542d7613b0e0d3c9e05e6b","score":58.272977800152034,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/automationbench-aa"},{"benchmarkId":"automationbench-aa","evidenceKind":"official_board","harnessId":"automationbench-aa-official","id":"s-automationbench-labs-devstral-small-2512-712aa4217411","ingestRunId":"ingest-automationbench-aa-712aa4217411","modelId":"labs-devstral-small-2512","n":657,"observedOn":"2026-09-29","rawPayloadHash":"712aa421741169313d1deaedce75a8ed04d9eede62542d7613b0e0d3c9e05e6b","score":3.0901842411547693,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/automationbench-aa"},{"benchmarkId":"automationbench-aa","evidenceKind":"official_board","harnessId":"automationbench-aa-official","id":"s-automationbench-llama-4-maverick-712aa4217411","ingestRunId":"ingest-automationbench-aa-712aa4217411","modelId":"llama-4-maverick","n":657,"observedOn":"2026-09-29","rawPayloadHash":"712aa421741169313d1deaedce75a8ed04d9eede62542d7613b0e0d3c9e05e6b","score":0.2121918776569101,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/automationbench-aa"},{"benchmarkId":"automationbench-aa","evidenceKind":"official_board","harnessId":"automationbench-aa-official","id":"s-automationbench-llama-4-scout-712aa4217411","ingestRunId":"ingest-automationbench-aa-712aa4217411","modelId":"llama-4-scout","n":657,"observedOn":"2026-09-29","rawPayloadHash":"712aa421741169313d1deaedce75a8ed04d9eede62542d7613b0e0d3c9e05e6b","score":0.19111706206154658,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/automationbench-aa"},{"benchmarkId":"automationbench-aa","evidenceKind":"official_board","harnessId":"automationbench-aa-official","id":"s-automationbench-magistral-small-2509-712aa4217411","ingestRunId":"ingest-automationbench-aa-712aa4217411","modelId":"magistral-small-2509","n":657,"observedOn":"2026-09-29","rawPayloadHash":"712aa421741169313d1deaedce75a8ed04d9eede62542d7613b0e0d3c9e05e6b","score":0.19111706206154658,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/automationbench-aa"},{"benchmarkId":"automationbench-aa","evidenceKind":"official_board","harnessId":"automationbench-aa-official","id":"s-automationbench-mimo-v2.5-712aa4217411","ingestRunId":"ingest-automationbench-aa-712aa4217411","modelId":"mimo-v2.5","n":657,"observedOn":"2026-09-29","rawPayloadHash":"712aa421741169313d1deaedce75a8ed04d9eede62542d7613b0e0d3c9e05e6b","score":18.42951233004315,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/automationbench-aa"},{"benchmarkId":"automationbench-aa","evidenceKind":"official_board","harnessId":"automationbench-aa-official","id":"s-automationbench-mimo-v2.5-pro-712aa4217411","ingestRunId":"ingest-automationbench-aa-712aa4217411","modelId":"mimo-v2.5-pro","n":657,"observedOn":"2026-09-29","rawPayloadHash":"712aa421741169313d1deaedce75a8ed04d9eede62542d7613b0e0d3c9e05e6b","score":13.703616508004313,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/automationbench-aa"},{"benchmarkId":"automationbench-aa","evidenceKind":"official_board","harnessId":"automationbench-aa-official","id":"s-automationbench-minicpm5-2b-712aa4217411","ingestRunId":"ingest-automationbench-aa-712aa4217411","modelId":"minicpm5-2b","n":657,"observedOn":"2026-09-29","rawPayloadHash":"712aa421741169313d1deaedce75a8ed04d9eede62542d7613b0e0d3c9e05e6b","score":1.8006007862431708,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/automationbench-aa"},{"benchmarkId":"automationbench-aa","evidenceKind":"official_board","harnessId":"automationbench-aa-official","id":"s-automationbench-minimax-m2.7-712aa4217411","ingestRunId":"ingest-automationbench-aa-712aa4217411","modelId":"minimax-m2.7","n":657,"observedOn":"2026-09-29","rawPayloadHash":"712aa421741169313d1deaedce75a8ed04d9eede62542d7613b0e0d3c9e05e6b","score":4.435290250372898,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/automationbench-aa"},{"benchmarkId":"automationbench-aa","evidenceKind":"official_board","harnessId":"automationbench-aa-official","id":"s-automationbench-minimax-m3-712aa4217411","ingestRunId":"ingest-automationbench-aa-712aa4217411","modelId":"minimax-m3","n":657,"observedOn":"2026-09-29","rawPayloadHash":"712aa421741169313d1deaedce75a8ed04d9eede62542d7613b0e0d3c9e05e6b","score":21.25125760074941,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/automationbench-aa"},{"benchmarkId":"automationbench-aa","evidenceKind":"official_board","harnessId":"automationbench-aa-official","id":"s-automationbench-ministral-3-14b-712aa4217411","ingestRunId":"ingest-automationbench-aa-712aa4217411","modelId":"ministral-3-14b","n":657,"observedOn":"2026-09-29","rawPayloadHash":"712aa421741169313d1deaedce75a8ed04d9eede62542d7613b0e0d3c9e05e6b","score":0.5512132673736124,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/automationbench-aa"},{"benchmarkId":"automationbench-aa","evidenceKind":"official_board","harnessId":"automationbench-aa-official","id":"s-automationbench-ministral-3-3b-712aa4217411","ingestRunId":"ingest-automationbench-aa-712aa4217411","modelId":"ministral-3-3b","n":657,"observedOn":"2026-09-29","rawPayloadHash":"712aa421741169313d1deaedce75a8ed04d9eede62542d7613b0e0d3c9e05e6b","score":0.6658657772309807,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/automationbench-aa"},{"benchmarkId":"automationbench-aa","evidenceKind":"official_board","harnessId":"automationbench-aa-official","id":"s-automationbench-ministral-3-8b-712aa4217411","ingestRunId":"ingest-automationbench-aa-712aa4217411","modelId":"ministral-3-8b","n":657,"observedOn":"2026-09-29","rawPayloadHash":"712aa421741169313d1deaedce75a8ed04d9eede62542d7613b0e0d3c9e05e6b","score":0.516749466666553,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/automationbench-aa"},{"benchmarkId":"automationbench-aa","evidenceKind":"official_board","harnessId":"automationbench-aa-official","id":"s-automationbench-mistral-large-3-712aa4217411","ingestRunId":"ingest-automationbench-aa-712aa4217411","modelId":"mistral-large-3","n":657,"observedOn":"2026-09-29","rawPayloadHash":"712aa421741169313d1deaedce75a8ed04d9eede62542d7613b0e0d3c9e05e6b","score":1.4974701708193463,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/automationbench-aa"},{"benchmarkId":"automationbench-aa","evidenceKind":"official_board","harnessId":"automationbench-aa-official","id":"s-automationbench-mistral-medium-2508-712aa4217411","ingestRunId":"ingest-automationbench-aa-712aa4217411","modelId":"mistral-medium-2508","n":657,"observedOn":"2026-09-29","rawPayloadHash":"712aa421741169313d1deaedce75a8ed04d9eede62542d7613b0e0d3c9e05e6b","score":4.146227762526245,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/automationbench-aa"},{"benchmarkId":"automationbench-aa","evidenceKind":"official_board","harnessId":"automationbench-aa-official","id":"s-automationbench-mistral-medium-3.5-712aa4217411","ingestRunId":"ingest-automationbench-aa-712aa4217411","modelId":"mistral-medium-3.5","n":657,"observedOn":"2026-09-29","rawPayloadHash":"712aa421741169313d1deaedce75a8ed04d9eede62542d7613b0e0d3c9e05e6b","score":6.299648656993127,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/automationbench-aa"},{"benchmarkId":"automationbench-aa","evidenceKind":"official_board","harnessId":"automationbench-aa-official","id":"s-automationbench-mistral-small-2506-712aa4217411","ingestRunId":"ingest-automationbench-aa-712aa4217411","modelId":"mistral-small-2506","n":657,"observedOn":"2026-09-29","rawPayloadHash":"712aa421741169313d1deaedce75a8ed04d9eede62542d7613b0e0d3c9e05e6b","score":0.2854023922677291,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/automationbench-aa"},{"benchmarkId":"automationbench-aa","evidenceKind":"official_board","harnessId":"automationbench-aa-official","id":"s-automationbench-mistral-small-4-712aa4217411","ingestRunId":"ingest-automationbench-aa-712aa4217411","modelId":"mistral-small-4","n":657,"observedOn":"2026-09-29","rawPayloadHash":"712aa421741169313d1deaedce75a8ed04d9eede62542d7613b0e0d3c9e05e6b","score":0.9635410038882158,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/automationbench-aa"},{"benchmarkId":"automationbench-aa","evidenceKind":"official_board","harnessId":"automationbench-aa-official","id":"s-automationbench-nemotron-3-ultra-712aa4217411","ingestRunId":"ingest-automationbench-aa-712aa4217411","modelId":"nemotron-3-ultra","n":657,"observedOn":"2026-09-29","rawPayloadHash":"712aa421741169313d1deaedce75a8ed04d9eede62542d7613b0e0d3c9e05e6b","score":2.986349541593992,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/automationbench-aa"},{"benchmarkId":"automationbench-aa","evidenceKind":"official_board","harnessId":"automationbench-aa-official","id":"s-automationbench-qwen-3.8-max-712aa4217411","ingestRunId":"ingest-automationbench-aa-712aa4217411","modelId":"qwen-3.8-max","n":657,"observedOn":"2026-09-29","rawPayloadHash":"712aa421741169313d1deaedce75a8ed04d9eede62542d7613b0e0d3c9e05e6b","score":56.1905381087678,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/automationbench-aa"},{"benchmarkId":"automationbench-aa","evidenceKind":"official_board","harnessId":"automationbench-aa-official","id":"s-automationbench-qwen3-coder-next-712aa4217411","ingestRunId":"ingest-automationbench-aa-712aa4217411","modelId":"qwen3-coder-next","n":657,"observedOn":"2026-09-29","rawPayloadHash":"712aa421741169313d1deaedce75a8ed04d9eede62542d7613b0e0d3c9e05e6b","score":1.0593550691766298,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/automationbench-aa"},{"benchmarkId":"automationbench-aa","evidenceKind":"official_board","harnessId":"automationbench-aa-official","id":"s-automationbench-qwen3.8-flash-next-712aa4217411","ingestRunId":"ingest-automationbench-aa-712aa4217411","modelId":"qwen3.8-flash-next","n":657,"observedOn":"2026-09-29","rawPayloadHash":"712aa421741169313d1deaedce75a8ed04d9eede62542d7613b0e0d3c9e05e6b","score":55.9083218951688,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/automationbench-aa"},{"benchmarkId":"automationbench-aa","evidenceKind":"official_board","harnessId":"automationbench-aa-official","id":"s-automationbench-solar-pro3-260323-712aa4217411","ingestRunId":"ingest-automationbench-aa-712aa4217411","modelId":"solar-pro3-260323","n":657,"observedOn":"2026-09-29","rawPayloadHash":"712aa421741169313d1deaedce75a8ed04d9eede62542d7613b0e0d3c9e05e6b","score":0.49734242022629577,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/automationbench-aa"},{"benchmarkId":"automationbench-aa","evidenceKind":"official_board","harnessId":"automationbench-aa-official","id":"s-automationbench-solar-pro4-260806-712aa4217411","ingestRunId":"ingest-automationbench-aa-712aa4217411","modelId":"solar-pro4-260806","n":657,"observedOn":"2026-09-29","rawPayloadHash":"712aa421741169313d1deaedce75a8ed04d9eede62542d7613b0e0d3c9e05e6b","score":9.082506839361528,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/automationbench-aa"},{"benchmarkId":"autoresearchexam","evidenceKind":"official_board","harnessId":"autoresearchexam-terminus-2","id":"s-autoresearchexam-claude-fable-5-1-3d089595005a","ingestRunId":"ingest-autoresearchexam-3d089595005a-fb878f2f","modelId":"claude-fable-5-1","n":29,"observedOn":"2026-09-09","rawPayloadHash":"3d089595005ac25718305d4daa79c9ba8461875c0247af563f7282fa8b43f72b","score":0.602,"scoreUnit":"index","sourceUrl":"https://benchmarks.bespokelabs.ai/autoresearchexam/"},{"benchmarkId":"autoresearchexam","evidenceKind":"official_board","harnessId":"autoresearchexam-terminus-2","id":"s-autoresearchexam-claude-opus-5-3d089595005a","ingestRunId":"ingest-autoresearchexam-3d089595005a-fb878f2f","modelId":"claude-opus-5","n":29,"observedOn":"2026-09-09","rawPayloadHash":"3d089595005ac25718305d4daa79c9ba8461875c0247af563f7282fa8b43f72b","score":0.579,"scoreUnit":"index","sourceUrl":"https://benchmarks.bespokelabs.ai/autoresearchexam/"},{"benchmarkId":"autoresearchexam","evidenceKind":"official_board","harnessId":"autoresearchexam-terminus-2","id":"s-autoresearchexam-gemini-3.8-flash-3d089595005a","ingestRunId":"ingest-autoresearchexam-3d089595005a-fb878f2f","modelId":"gemini-3.8-flash","n":29,"observedOn":"2026-09-09","rawPayloadHash":"3d089595005ac25718305d4daa79c9ba8461875c0247af563f7282fa8b43f72b","score":0.484,"scoreUnit":"index","sourceUrl":"https://benchmarks.bespokelabs.ai/autoresearchexam/"},{"benchmarkId":"autoresearchexam","evidenceKind":"official_board","harnessId":"autoresearchexam-terminus-2","id":"s-autoresearchexam-gpt-5.6-sol-3d089595005a","ingestRunId":"ingest-autoresearchexam-3d089595005a-fb878f2f","modelId":"gpt-5.6-sol","n":29,"observedOn":"2026-09-09","rawPayloadHash":"3d089595005ac25718305d4daa79c9ba8461875c0247af563f7282fa8b43f72b","score":0.513,"scoreUnit":"index","sourceUrl":"https://benchmarks.bespokelabs.ai/autoresearchexam/"},{"benchmarkId":"autoresearchexam","evidenceKind":"official_board","harnessId":"autoresearchexam-terminus-2","id":"s-autoresearchexam-gpt-6-astra-3d089595005a","ingestRunId":"ingest-autoresearchexam-3d089595005a-fb878f2f","modelId":"gpt-6-astra","n":29,"observedOn":"2026-09-09","rawPayloadHash":"3d089595005ac25718305d4daa79c9ba8461875c0247af563f7282fa8b43f72b","score":0.6,"scoreUnit":"index","sourceUrl":"https://benchmarks.bespokelabs.ai/autoresearchexam/"},{"benchmarkId":"autoresearchexam","evidenceKind":"official_board","harnessId":"autoresearchexam-terminus-2","id":"s-autoresearchexam-grok-4.6-3d089595005a","ingestRunId":"ingest-autoresearchexam-3d089595005a-fb878f2f","modelId":"grok-4.6","n":29,"observedOn":"2026-09-09","rawPayloadHash":"3d089595005ac25718305d4daa79c9ba8461875c0247af563f7282fa8b43f72b","score":0.493,"scoreUnit":"index","sourceUrl":"https://benchmarks.bespokelabs.ai/autoresearchexam/"},{"benchmarkId":"autoresearchexam","evidenceKind":"official_board","harnessId":"autoresearchexam-terminus-2","id":"s-autoresearchexam-kimi-k3-3d089595005a","ingestRunId":"ingest-autoresearchexam-3d089595005a-fb878f2f","modelId":"kimi-k3","n":29,"observedOn":"2026-09-09","rawPayloadHash":"3d089595005ac25718305d4daa79c9ba8461875c0247af563f7282fa8b43f72b","score":0.435,"scoreUnit":"index","sourceUrl":"https://benchmarks.bespokelabs.ai/autoresearchexam/"},{"benchmarkId":"autoresearchexam","evidenceKind":"official_board","harnessId":"autoresearchexam-terminus-2","id":"s-autoresearchexam-muse-spark-1.3-3d089595005a","ingestRunId":"ingest-autoresearchexam-3d089595005a-fb878f2f","modelId":"muse-spark-1.3","n":29,"observedOn":"2026-09-09","rawPayloadHash":"3d089595005ac25718305d4daa79c9ba8461875c0247af563f7282fa8b43f72b","score":0.379,"scoreUnit":"index","sourceUrl":"https://benchmarks.bespokelabs.ai/autoresearchexam/"},{"benchmarkId":"autoresearchexam","evidenceKind":"official_board","harnessId":"autoresearchexam-terminus-2","id":"s-autoresearchexam-qwen-3.8-max-3d089595005a","ingestRunId":"ingest-autoresearchexam-3d089595005a-fb878f2f","modelId":"qwen-3.8-max","n":29,"observedOn":"2026-09-09","rawPayloadHash":"3d089595005ac25718305d4daa79c9ba8461875c0247af563f7282fa8b43f72b","score":0.423,"scoreUnit":"index","sourceUrl":"https://benchmarks.bespokelabs.ai/autoresearchexam/"},{"benchmarkId":"bughunt-bench","evidenceKind":"official_board","harnessId":"bughunt-bench-high","id":"s-bughunt-claude-fable-5-1-high-e184b9c0609a","ingestRunId":"ingest-bughunt-bench-e184b9c0609a-fb878f2f","modelId":"claude-fable-5-1","n":105,"observedOn":"2026-09-01","rawPayloadHash":"e184b9c0609adaffbbdf2762b3e9b66329c9a7d1d3856ca98fd1972a714ff56a","score":31.43,"scoreUnit":"percent","sourceUrl":"https://bughunt.productcompass.pm/?preset=featured"},{"benchmarkId":"bughunt-bench","evidenceKind":"official_board","harnessId":"bughunt-bench-max","id":"s-bughunt-claude-fable-5-1-max-e184b9c0609a","ingestRunId":"ingest-bughunt-bench-e184b9c0609a-fb878f2f","modelId":"claude-fable-5-1","n":105,"observedOn":"2026-09-01","rawPayloadHash":"e184b9c0609adaffbbdf2762b3e9b66329c9a7d1d3856ca98fd1972a714ff56a","score":40.95,"scoreUnit":"percent","sourceUrl":"https://bughunt.productcompass.pm/?preset=featured"},{"benchmarkId":"bughunt-bench","evidenceKind":"official_board","harnessId":"bughunt-bench-high","id":"s-bughunt-claude-opus-5-5-high-e184b9c0609a","ingestRunId":"ingest-bughunt-bench-e184b9c0609a-fb878f2f","modelId":"claude-opus-5-5","n":105,"observedOn":"2026-09-23","rawPayloadHash":"e184b9c0609adaffbbdf2762b3e9b66329c9a7d1d3856ca98fd1972a714ff56a","score":30.19,"scoreUnit":"percent","sourceUrl":"https://bughunt.productcompass.pm/?preset=featured"},{"benchmarkId":"bughunt-bench","evidenceKind":"official_board","harnessId":"bughunt-bench-max","id":"s-bughunt-claude-opus-5-5-max-e184b9c0609a","ingestRunId":"ingest-bughunt-bench-e184b9c0609a-fb878f2f","modelId":"claude-opus-5-5","n":105,"observedOn":"2026-09-23","rawPayloadHash":"e184b9c0609adaffbbdf2762b3e9b66329c9a7d1d3856ca98fd1972a714ff56a","score":39.71,"scoreUnit":"percent","sourceUrl":"https://bughunt.productcompass.pm/?preset=featured"},{"benchmarkId":"bughunt-bench","evidenceKind":"official_board","harnessId":"bughunt-bench-xhigh","id":"s-bughunt-claude-opus-5-5-xhigh-e184b9c0609a","ingestRunId":"ingest-bughunt-bench-e184b9c0609a-fb878f2f","modelId":"claude-opus-5-5","n":105,"observedOn":"2026-09-23","rawPayloadHash":"e184b9c0609adaffbbdf2762b3e9b66329c9a7d1d3856ca98fd1972a714ff56a","score":34.29,"scoreUnit":"percent","sourceUrl":"https://bughunt.productcompass.pm/?preset=featured"},{"benchmarkId":"bughunt-bench","evidenceKind":"official_board","harnessId":"bughunt-bench-high","id":"s-bughunt-claude-opus-5-high-3b0ee33cbf7f","ingestRunId":"ingest-bughunt-bench-3b0ee33cbf7f-f55dfe2b","modelId":"claude-opus-5","n":105,"observedOn":"2026-07-26","rawPayloadHash":"3b0ee33cbf7f7faf824af5d984dc3a18f418567d24b6a9ffde892882d88dab44","score":20,"scoreUnit":"percent","sourceUrl":"https://bughunt.productcompass.pm/?preset=featured"},{"benchmarkId":"bughunt-bench","evidenceKind":"official_board","harnessId":"bughunt-bench-max","id":"s-bughunt-claude-opus-5-max-e184b9c0609a","ingestRunId":"ingest-bughunt-bench-e184b9c0609a-fb878f2f","modelId":"claude-opus-5","n":105,"observedOn":"2026-08-01","rawPayloadHash":"e184b9c0609adaffbbdf2762b3e9b66329c9a7d1d3856ca98fd1972a714ff56a","score":25.71,"scoreUnit":"percent","sourceUrl":"https://bughunt.productcompass.pm/?preset=featured"},{"benchmarkId":"bughunt-bench","evidenceKind":"official_board","harnessId":"bughunt-bench-max","id":"s-bughunt-claude-sonnet-5-5-max-e184b9c0609a","ingestRunId":"ingest-bughunt-bench-e184b9c0609a-fb878f2f","modelId":"claude-sonnet-5-5","n":105,"observedOn":"2026-09-29","rawPayloadHash":"e184b9c0609adaffbbdf2762b3e9b66329c9a7d1d3856ca98fd1972a714ff56a","score":48.86,"scoreUnit":"percent","sourceUrl":"https://bughunt.productcompass.pm/?preset=featured"},{"benchmarkId":"bughunt-bench","evidenceKind":"official_board","harnessId":"bughunt-bench-high","id":"s-bughunt-gemini-3.8-flash-high-e184b9c0609a","ingestRunId":"ingest-bughunt-bench-e184b9c0609a-fb878f2f","modelId":"gemini-3.8-flash","n":105,"observedOn":"2026-09-15","rawPayloadHash":"e184b9c0609adaffbbdf2762b3e9b66329c9a7d1d3856ca98fd1972a714ff56a","score":17.14,"scoreUnit":"percent","sourceUrl":"https://bughunt.productcompass.pm/?preset=featured"},{"benchmarkId":"bughunt-bench","evidenceKind":"official_board","harnessId":"bughunt-bench-max","id":"s-bughunt-glm-5.3-flash-max-e184b9c0609a","ingestRunId":"ingest-bughunt-bench-e184b9c0609a-fb878f2f","modelId":"glm-5.3-flash","n":105,"observedOn":"2026-09-15","rawPayloadHash":"e184b9c0609adaffbbdf2762b3e9b66329c9a7d1d3856ca98fd1972a714ff56a","score":16.86,"scoreUnit":"percent","sourceUrl":"https://bughunt.productcompass.pm/?preset=featured"},{"benchmarkId":"bughunt-bench","evidenceKind":"official_board","harnessId":"bughunt-bench-max","id":"s-bughunt-glm-5.3-max-e184b9c0609a","ingestRunId":"ingest-bughunt-bench-e184b9c0609a-fb878f2f","modelId":"glm-5.3","n":105,"observedOn":"2026-09-10","rawPayloadHash":"e184b9c0609adaffbbdf2762b3e9b66329c9a7d1d3856ca98fd1972a714ff56a","score":18.1,"scoreUnit":"percent","sourceUrl":"https://bughunt.productcompass.pm/?preset=featured"},{"benchmarkId":"bughunt-bench","evidenceKind":"official_board","harnessId":"bughunt-bench-max","id":"s-bughunt-gpt-5.6-luna-max-e184b9c0609a","ingestRunId":"ingest-bughunt-bench-e184b9c0609a-fb878f2f","modelId":"gpt-5.6-luna","n":105,"observedOn":"2026-09-22","rawPayloadHash":"e184b9c0609adaffbbdf2762b3e9b66329c9a7d1d3856ca98fd1972a714ff56a","score":29.81,"scoreUnit":"percent","sourceUrl":"https://bughunt.productcompass.pm/?preset=featured"},{"benchmarkId":"bughunt-bench","evidenceKind":"official_board","harnessId":"bughunt-bench-high","id":"s-bughunt-gpt-5.6-sol-high-3b0ee33cbf7f","ingestRunId":"ingest-bughunt-bench-3b0ee33cbf7f-f55dfe2b","modelId":"gpt-5.6-sol","n":105,"observedOn":"2026-07-31","rawPayloadHash":"3b0ee33cbf7f7faf824af5d984dc3a18f418567d24b6a9ffde892882d88dab44","score":32.38,"scoreUnit":"percent","sourceUrl":"https://bughunt.productcompass.pm/?preset=featured"},{"benchmarkId":"bughunt-bench","evidenceKind":"official_board","harnessId":"bughunt-bench-max","id":"s-bughunt-gpt-5.6-sol-max-e184b9c0609a","ingestRunId":"ingest-bughunt-bench-e184b9c0609a-fb878f2f","modelId":"gpt-5.6-sol","n":105,"observedOn":"2026-09-16","rawPayloadHash":"e184b9c0609adaffbbdf2762b3e9b66329c9a7d1d3856ca98fd1972a714ff56a","score":41.43,"scoreUnit":"percent","sourceUrl":"https://bughunt.productcompass.pm/?preset=featured"},{"benchmarkId":"bughunt-bench","evidenceKind":"official_board","harnessId":"bughunt-bench-max","id":"s-bughunt-gpt-6-astra-max-e184b9c0609a","ingestRunId":"ingest-bughunt-bench-e184b9c0609a-fb878f2f","modelId":"gpt-6-astra","n":105,"observedOn":"2026-09-14","rawPayloadHash":"e184b9c0609adaffbbdf2762b3e9b66329c9a7d1d3856ca98fd1972a714ff56a","score":42.86,"scoreUnit":"percent","sourceUrl":"https://bughunt.productcompass.pm/?preset=featured"},{"benchmarkId":"bughunt-bench","evidenceKind":"official_board","harnessId":"bughunt-bench-xhigh","id":"s-bughunt-gpt-6-astra-xhigh-e184b9c0609a","ingestRunId":"ingest-bughunt-bench-e184b9c0609a-fb878f2f","modelId":"gpt-6-astra","n":105,"observedOn":"2026-09-04","rawPayloadHash":"e184b9c0609adaffbbdf2762b3e9b66329c9a7d1d3856ca98fd1972a714ff56a","score":40.95,"scoreUnit":"percent","sourceUrl":"https://bughunt.productcompass.pm/?preset=featured"},{"benchmarkId":"bughunt-bench","evidenceKind":"official_board","harnessId":"bughunt-bench-max","id":"s-bughunt-gpt-6-luna-max-e184b9c0609a","ingestRunId":"ingest-bughunt-bench-e184b9c0609a-fb878f2f","modelId":"gpt-6-luna","n":105,"observedOn":"2026-09-23","rawPayloadHash":"e184b9c0609adaffbbdf2762b3e9b66329c9a7d1d3856ca98fd1972a714ff56a","score":17.43,"scoreUnit":"percent","sourceUrl":"https://bughunt.productcompass.pm/?preset=featured"},{"benchmarkId":"bughunt-bench","evidenceKind":"official_board","harnessId":"bughunt-bench-high","id":"s-bughunt-gpt-6-sol-high-e184b9c0609a","ingestRunId":"ingest-bughunt-bench-e184b9c0609a-fb878f2f","modelId":"gpt-6-sol","n":105,"observedOn":"2026-09-23","rawPayloadHash":"e184b9c0609adaffbbdf2762b3e9b66329c9a7d1d3856ca98fd1972a714ff56a","score":19.05,"scoreUnit":"percent","sourceUrl":"https://bughunt.productcompass.pm/?preset=featured"},{"benchmarkId":"bughunt-bench","evidenceKind":"official_board","harnessId":"bughunt-bench-max","id":"s-bughunt-gpt-6-sol-max-e184b9c0609a","ingestRunId":"ingest-bughunt-bench-e184b9c0609a-fb878f2f","modelId":"gpt-6-sol","n":105,"observedOn":"2026-09-23","rawPayloadHash":"e184b9c0609adaffbbdf2762b3e9b66329c9a7d1d3856ca98fd1972a714ff56a","score":27.9,"scoreUnit":"percent","sourceUrl":"https://bughunt.productcompass.pm/?preset=featured"},{"benchmarkId":"bughunt-bench","evidenceKind":"official_board","harnessId":"bughunt-bench-xhigh","id":"s-bughunt-gpt-6-sol-xhigh-e184b9c0609a","ingestRunId":"ingest-bughunt-bench-e184b9c0609a-fb878f2f","modelId":"gpt-6-sol","n":105,"observedOn":"2026-09-23","rawPayloadHash":"e184b9c0609adaffbbdf2762b3e9b66329c9a7d1d3856ca98fd1972a714ff56a","score":23.81,"scoreUnit":"percent","sourceUrl":"https://bughunt.productcompass.pm/?preset=featured"},{"benchmarkId":"bughunt-bench","evidenceKind":"official_board","harnessId":"bughunt-bench-max","id":"s-bughunt-gpt-6.1-sol-max-e184b9c0609a","ingestRunId":"ingest-bughunt-bench-e184b9c0609a-fb878f2f","modelId":"gpt-6.1-sol","n":105,"observedOn":"2026-09-29","rawPayloadHash":"e184b9c0609adaffbbdf2762b3e9b66329c9a7d1d3856ca98fd1972a714ff56a","score":41.9,"scoreUnit":"percent","sourceUrl":"https://bughunt.productcompass.pm/?preset=featured"},{"benchmarkId":"bughunt-bench","evidenceKind":"official_board","harnessId":"bughunt-bench-xhigh","id":"s-bughunt-grok-4.6-xhigh-e184b9c0609a","ingestRunId":"ingest-bughunt-bench-e184b9c0609a-fb878f2f","modelId":"grok-4.6","n":105,"observedOn":"2026-09-14","rawPayloadHash":"e184b9c0609adaffbbdf2762b3e9b66329c9a7d1d3856ca98fd1972a714ff56a","score":27.33,"scoreUnit":"percent","sourceUrl":"https://bughunt.productcompass.pm/?preset=featured"},{"benchmarkId":"bughunt-bench","evidenceKind":"official_board","harnessId":"bughunt-bench-xhigh","id":"s-bughunt-grok-4.7-xhigh-e184b9c0609a","ingestRunId":"ingest-bughunt-bench-e184b9c0609a-fb878f2f","modelId":"grok-4.7","n":105,"observedOn":"2026-09-21","rawPayloadHash":"e184b9c0609adaffbbdf2762b3e9b66329c9a7d1d3856ca98fd1972a714ff56a","score":27.43,"scoreUnit":"percent","sourceUrl":"https://bughunt.productcompass.pm/?preset=featured"},{"benchmarkId":"bughunt-bench","evidenceKind":"official_board","harnessId":"bughunt-bench-default","id":"s-bughunt-kimi-k3-default-e184b9c0609a","ingestRunId":"ingest-bughunt-bench-e184b9c0609a-fb878f2f","modelId":"kimi-k3","n":105,"observedOn":"2026-07-26","rawPayloadHash":"e184b9c0609adaffbbdf2762b3e9b66329c9a7d1d3856ca98fd1972a714ff56a","score":20,"scoreUnit":"percent","sourceUrl":"https://bughunt.productcompass.pm/?preset=featured"},{"benchmarkId":"bughunt-bench","evidenceKind":"official_board","harnessId":"bughunt-bench-max","id":"s-bughunt-muse-spark-1.3-max-e184b9c0609a","ingestRunId":"ingest-bughunt-bench-e184b9c0609a-fb878f2f","modelId":"muse-spark-1.3","n":105,"observedOn":"2026-09-17","rawPayloadHash":"e184b9c0609adaffbbdf2762b3e9b66329c9a7d1d3856ca98fd1972a714ff56a","score":30.67,"scoreUnit":"percent","sourceUrl":"https://bughunt.productcompass.pm/?preset=featured"},{"benchmarkId":"bughunt-bench","evidenceKind":"official_board","harnessId":"bughunt-bench-xhigh","id":"s-bughunt-muse-spark-1.3-xhigh-e184b9c0609a","ingestRunId":"ingest-bughunt-bench-e184b9c0609a-fb878f2f","modelId":"muse-spark-1.3","n":105,"observedOn":"2026-09-14","rawPayloadHash":"e184b9c0609adaffbbdf2762b3e9b66329c9a7d1d3856ca98fd1972a714ff56a","score":19.33,"scoreUnit":"percent","sourceUrl":"https://bughunt.productcompass.pm/?preset=featured"},{"benchmarkId":"bughunt-bench","evidenceKind":"official_board","harnessId":"bughunt-bench-max","id":"s-bughunt-qwen-3.8-max-max-e184b9c0609a","ingestRunId":"ingest-bughunt-bench-e184b9c0609a-fb878f2f","modelId":"qwen-3.8-max","n":105,"observedOn":"2026-09-15","rawPayloadHash":"e184b9c0609adaffbbdf2762b3e9b66329c9a7d1d3856ca98fd1972a714ff56a","score":24.48,"scoreUnit":"percent","sourceUrl":"https://bughunt.productcompass.pm/?preset=featured"},{"benchmarkId":"bughunt-bench","evidenceKind":"official_board","harnessId":"bughunt-bench-xhigh","id":"s-bughunt-qwen3.8-27b-xhigh-e184b9c0609a","ingestRunId":"ingest-bughunt-bench-e184b9c0609a-fb878f2f","modelId":"qwen3.8-27b","n":105,"observedOn":"2026-09-15","rawPayloadHash":"e184b9c0609adaffbbdf2762b3e9b66329c9a7d1d3856ca98fd1972a714ff56a","score":14.29,"scoreUnit":"percent","sourceUrl":"https://bughunt.productcompass.pm/?preset=featured"},{"benchmarkId":"deepswe-v1.1","evidenceKind":"official_board","harnessId":"deepswe-v1.1-reported","id":"s-deepswe-claude-fable-5-eb88d1c756c0","ingestRunId":"ingest-deepswe-v1.1-eb88d1c756c0","modelId":"claude-fable-5","n":113,"observedOn":"2026-09-22","rawPayloadHash":"eb88d1c756c0ea2ad7ff534dbf9c4ebd36ed73e4ca1b560533ba8ae447f75ee4","score":69.9,"scoreUnit":"percent","sourceUrl":"https://deepswe.datacurve.ai/"},{"benchmarkId":"deepswe-v1.1","evidenceKind":"official_board","harnessId":"deepswe-v1.1-reported","id":"s-deepswe-claude-opus-4-8-eb88d1c756c0","ingestRunId":"ingest-deepswe-v1.1-eb88d1c756c0","modelId":"claude-opus-4-8","n":111,"observedOn":"2026-09-22","rawPayloadHash":"eb88d1c756c0ea2ad7ff534dbf9c4ebd36ed73e4ca1b560533ba8ae447f75ee4","score":59,"scoreUnit":"percent","sourceUrl":"https://deepswe.datacurve.ai/"},{"benchmarkId":"deepswe-v1.1","evidenceKind":"official_board","harnessId":"deepswe-v1.1-reported","id":"s-deepswe-claude-opus-5-eb88d1c756c0","ingestRunId":"ingest-deepswe-v1.1-eb88d1c756c0","modelId":"claude-opus-5","n":113,"observedOn":"2026-09-22","rawPayloadHash":"eb88d1c756c0ea2ad7ff534dbf9c4ebd36ed73e4ca1b560533ba8ae447f75ee4","score":73.6,"scoreUnit":"percent","sourceUrl":"https://deepswe.datacurve.ai/"},{"benchmarkId":"deepswe-v1.1","evidenceKind":"official_board","harnessId":"deepswe-v1.1-reported","id":"s-deepswe-claude-sonnet-4-6-eb88d1c756c0","ingestRunId":"ingest-deepswe-v1.1-eb88d1c756c0","modelId":"claude-sonnet-4-6","n":113,"observedOn":"2026-09-22","rawPayloadHash":"eb88d1c756c0ea2ad7ff534dbf9c4ebd36ed73e4ca1b560533ba8ae447f75ee4","score":29.9,"scoreUnit":"percent","sourceUrl":"https://deepswe.datacurve.ai/"},{"benchmarkId":"deepswe-v1.1","evidenceKind":"official_board","harnessId":"deepswe-v1.1-reported","id":"s-deepswe-claude-sonnet-5-eb88d1c756c0","ingestRunId":"ingest-deepswe-v1.1-eb88d1c756c0","modelId":"claude-sonnet-5","n":113,"observedOn":"2026-09-22","rawPayloadHash":"eb88d1c756c0ea2ad7ff534dbf9c4ebd36ed73e4ca1b560533ba8ae447f75ee4","score":53.8,"scoreUnit":"percent","sourceUrl":"https://deepswe.datacurve.ai/"},{"benchmarkId":"deepswe-v1.1","evidenceKind":"official_board","harnessId":"deepswe-v1.1-reported","id":"s-deepswe-deepseek-v4-flash-eb88d1c756c0","ingestRunId":"ingest-deepswe-v1.1-eb88d1c756c0","modelId":"deepseek-v4-flash","n":113,"observedOn":"2026-09-22","rawPayloadHash":"eb88d1c756c0ea2ad7ff534dbf9c4ebd36ed73e4ca1b560533ba8ae447f75ee4","score":53.3,"scoreUnit":"percent","sourceUrl":"https://deepswe.datacurve.ai/"},{"benchmarkId":"deepswe-v1.1","evidenceKind":"official_board","harnessId":"deepswe-v1.1-reported","id":"s-deepswe-deepseek-v4-pro-eb88d1c756c0","ingestRunId":"ingest-deepswe-v1.1-eb88d1c756c0","modelId":"deepseek-v4-pro","n":113,"observedOn":"2026-09-22","rawPayloadHash":"eb88d1c756c0ea2ad7ff534dbf9c4ebd36ed73e4ca1b560533ba8ae447f75ee4","score":62.8,"scoreUnit":"percent","sourceUrl":"https://deepswe.datacurve.ai/"},{"benchmarkId":"deepswe-v1.1","evidenceKind":"official_board","harnessId":"deepswe-v1.1-reported","id":"s-deepswe-gemini-3.1-pro-eb88d1c756c0","ingestRunId":"ingest-deepswe-v1.1-eb88d1c756c0","modelId":"gemini-3.1-pro","n":113,"observedOn":"2026-09-22","rawPayloadHash":"eb88d1c756c0ea2ad7ff534dbf9c4ebd36ed73e4ca1b560533ba8ae447f75ee4","score":11.7,"scoreUnit":"percent","sourceUrl":"https://deepswe.datacurve.ai/"},{"benchmarkId":"deepswe-v1.1","evidenceKind":"official_board","harnessId":"deepswe-v1.1-reported","id":"s-deepswe-gemini-3.5-flash-eb88d1c756c0","ingestRunId":"ingest-deepswe-v1.1-eb88d1c756c0","modelId":"gemini-3.5-flash","n":113,"observedOn":"2026-09-22","rawPayloadHash":"eb88d1c756c0ea2ad7ff534dbf9c4ebd36ed73e4ca1b560533ba8ae447f75ee4","score":36.1,"scoreUnit":"percent","sourceUrl":"https://deepswe.datacurve.ai/"},{"benchmarkId":"deepswe-v1.1","evidenceKind":"official_board","harnessId":"deepswe-v1.1-reported","id":"s-deepswe-gemini-3.6-flash-eb88d1c756c0","ingestRunId":"ingest-deepswe-v1.1-eb88d1c756c0","modelId":"gemini-3.6-flash","n":113,"observedOn":"2026-09-22","rawPayloadHash":"eb88d1c756c0ea2ad7ff534dbf9c4ebd36ed73e4ca1b560533ba8ae447f75ee4","score":46.7,"scoreUnit":"percent","sourceUrl":"https://deepswe.datacurve.ai/"},{"benchmarkId":"deepswe-v1.1","evidenceKind":"official_board","harnessId":"deepswe-v1.1-reported","id":"s-deepswe-gemini-3.7-flash-eb88d1c756c0","ingestRunId":"ingest-deepswe-v1.1-eb88d1c756c0","modelId":"gemini-3.7-flash","n":113,"observedOn":"2026-09-22","rawPayloadHash":"eb88d1c756c0ea2ad7ff534dbf9c4ebd36ed73e4ca1b560533ba8ae447f75ee4","score":65.5,"scoreUnit":"percent","sourceUrl":"https://deepswe.datacurve.ai/"},{"benchmarkId":"deepswe-v1.1","evidenceKind":"official_board","harnessId":"deepswe-v1.1-reported","id":"s-deepswe-gemini-3.8-flash-eb88d1c756c0","ingestRunId":"ingest-deepswe-v1.1-eb88d1c756c0","modelId":"gemini-3.8-flash","n":113,"observedOn":"2026-09-22","rawPayloadHash":"eb88d1c756c0ea2ad7ff534dbf9c4ebd36ed73e4ca1b560533ba8ae447f75ee4","score":73.8,"scoreUnit":"percent","sourceUrl":"https://deepswe.datacurve.ai/"},{"benchmarkId":"deepswe-v1.1","evidenceKind":"official_board","harnessId":"deepswe-v1.1-reported","id":"s-deepswe-glm-5.2-eb88d1c756c0","ingestRunId":"ingest-deepswe-v1.1-eb88d1c756c0","modelId":"glm-5.2","n":113,"observedOn":"2026-09-22","rawPayloadHash":"eb88d1c756c0ea2ad7ff534dbf9c4ebd36ed73e4ca1b560533ba8ae447f75ee4","score":43.8,"scoreUnit":"percent","sourceUrl":"https://deepswe.datacurve.ai/"},{"benchmarkId":"deepswe-v1.1","evidenceKind":"official_board","harnessId":"deepswe-v1.1-reported","id":"s-deepswe-glm-5.3-eb88d1c756c0","ingestRunId":"ingest-deepswe-v1.1-eb88d1c756c0","modelId":"glm-5.3","n":113,"observedOn":"2026-09-22","rawPayloadHash":"eb88d1c756c0ea2ad7ff534dbf9c4ebd36ed73e4ca1b560533ba8ae447f75ee4","score":69,"scoreUnit":"percent","sourceUrl":"https://deepswe.datacurve.ai/"},{"benchmarkId":"deepswe-v1.1","evidenceKind":"official_board","harnessId":"deepswe-v1.1-reported","id":"s-deepswe-glm-5.3-flash-eb88d1c756c0","ingestRunId":"ingest-deepswe-v1.1-eb88d1c756c0","modelId":"glm-5.3-flash","n":113,"observedOn":"2026-09-22","rawPayloadHash":"eb88d1c756c0ea2ad7ff534dbf9c4ebd36ed73e4ca1b560533ba8ae447f75ee4","score":63.4,"scoreUnit":"percent","sourceUrl":"https://deepswe.datacurve.ai/"},{"benchmarkId":"deepswe-v1.1","evidenceKind":"official_board","harnessId":"deepswe-v1.1-reported","id":"s-deepswe-gpt-5.4-eb88d1c756c0","ingestRunId":"ingest-deepswe-v1.1-eb88d1c756c0","modelId":"gpt-5.4","n":113,"observedOn":"2026-09-22","rawPayloadHash":"eb88d1c756c0ea2ad7ff534dbf9c4ebd36ed73e4ca1b560533ba8ae447f75ee4","score":51.8,"scoreUnit":"percent","sourceUrl":"https://deepswe.datacurve.ai/"},{"benchmarkId":"deepswe-v1.1","evidenceKind":"official_board","harnessId":"deepswe-v1.1-reported","id":"s-deepswe-gpt-5.5-eb88d1c756c0","ingestRunId":"ingest-deepswe-v1.1-eb88d1c756c0","modelId":"gpt-5.5","n":113,"observedOn":"2026-09-22","rawPayloadHash":"eb88d1c756c0ea2ad7ff534dbf9c4ebd36ed73e4ca1b560533ba8ae447f75ee4","score":67,"scoreUnit":"percent","sourceUrl":"https://deepswe.datacurve.ai/"},{"benchmarkId":"deepswe-v1.1","evidenceKind":"official_board","harnessId":"deepswe-v1.1-reported","id":"s-deepswe-gpt-5.6-luna-eb88d1c756c0","ingestRunId":"ingest-deepswe-v1.1-eb88d1c756c0","modelId":"gpt-5.6-luna","n":113,"observedOn":"2026-09-22","rawPayloadHash":"eb88d1c756c0ea2ad7ff534dbf9c4ebd36ed73e4ca1b560533ba8ae447f75ee4","score":67.2,"scoreUnit":"percent","sourceUrl":"https://deepswe.datacurve.ai/"},{"benchmarkId":"deepswe-v1.1","evidenceKind":"official_board","harnessId":"deepswe-v1.1-reported","id":"s-deepswe-gpt-5.6-sol-eb88d1c756c0","ingestRunId":"ingest-deepswe-v1.1-eb88d1c756c0","modelId":"gpt-5.6-sol","n":113,"observedOn":"2026-09-22","rawPayloadHash":"eb88d1c756c0ea2ad7ff534dbf9c4ebd36ed73e4ca1b560533ba8ae447f75ee4","score":72.7,"scoreUnit":"percent","sourceUrl":"https://deepswe.datacurve.ai/"},{"benchmarkId":"deepswe-v1.1","evidenceKind":"official_board","harnessId":"deepswe-v1.1-reported","id":"s-deepswe-gpt-5.6-terra-eb88d1c756c0","ingestRunId":"ingest-deepswe-v1.1-eb88d1c756c0","modelId":"gpt-5.6-terra","n":113,"observedOn":"2026-09-22","rawPayloadHash":"eb88d1c756c0ea2ad7ff534dbf9c4ebd36ed73e4ca1b560533ba8ae447f75ee4","score":69.6,"scoreUnit":"percent","sourceUrl":"https://deepswe.datacurve.ai/"},{"benchmarkId":"deepswe-v1.1","evidenceKind":"official_board","harnessId":"deepswe-v1.1-reported","id":"s-deepswe-gpt-6-astra-eb88d1c756c0","ingestRunId":"ingest-deepswe-v1.1-eb88d1c756c0","modelId":"gpt-6-astra","n":113,"observedOn":"2026-09-22","rawPayloadHash":"eb88d1c756c0ea2ad7ff534dbf9c4ebd36ed73e4ca1b560533ba8ae447f75ee4","score":74.1,"scoreUnit":"percent","sourceUrl":"https://deepswe.datacurve.ai/"},{"benchmarkId":"deepswe-v1.1","evidenceKind":"official_board","harnessId":"deepswe-v1.1-reported","id":"s-deepswe-grok-4.5-eb88d1c756c0","ingestRunId":"ingest-deepswe-v1.1-eb88d1c756c0","modelId":"grok-4.5","n":113,"observedOn":"2026-09-22","rawPayloadHash":"eb88d1c756c0ea2ad7ff534dbf9c4ebd36ed73e4ca1b560533ba8ae447f75ee4","score":53.8,"scoreUnit":"percent","sourceUrl":"https://deepswe.datacurve.ai/"},{"benchmarkId":"deepswe-v1.1","evidenceKind":"official_board","harnessId":"deepswe-v1.1-reported","id":"s-deepswe-grok-4.6-eb88d1c756c0","ingestRunId":"ingest-deepswe-v1.1-eb88d1c756c0","modelId":"grok-4.6","n":113,"observedOn":"2026-09-22","rawPayloadHash":"eb88d1c756c0ea2ad7ff534dbf9c4ebd36ed73e4ca1b560533ba8ae447f75ee4","score":67.5,"scoreUnit":"percent","sourceUrl":"https://deepswe.datacurve.ai/"},{"benchmarkId":"deepswe-v1.1","evidenceKind":"official_board","harnessId":"deepswe-v1.1-reported","id":"s-deepswe-kimi-k2.7-code-eb88d1c756c0","ingestRunId":"ingest-deepswe-v1.1-eb88d1c756c0","modelId":"kimi-k2.7-code","n":113,"observedOn":"2026-09-22","rawPayloadHash":"eb88d1c756c0ea2ad7ff534dbf9c4ebd36ed73e4ca1b560533ba8ae447f75ee4","score":30.5,"scoreUnit":"percent","sourceUrl":"https://deepswe.datacurve.ai/"},{"benchmarkId":"deepswe-v1.1","evidenceKind":"official_board","harnessId":"deepswe-v1.1-reported","id":"s-deepswe-kimi-k3-eb88d1c756c0","ingestRunId":"ingest-deepswe-v1.1-eb88d1c756c0","modelId":"kimi-k3","n":113,"observedOn":"2026-09-22","rawPayloadHash":"eb88d1c756c0ea2ad7ff534dbf9c4ebd36ed73e4ca1b560533ba8ae447f75ee4","score":68.5,"scoreUnit":"percent","sourceUrl":"https://deepswe.datacurve.ai/"},{"benchmarkId":"deepswe-v1.1","evidenceKind":"official_board","harnessId":"deepswe-v1.1-reported","id":"s-deepswe-muse-spark-1.1-eb88d1c756c0","ingestRunId":"ingest-deepswe-v1.1-eb88d1c756c0","modelId":"muse-spark-1.1","n":113,"observedOn":"2026-09-22","rawPayloadHash":"eb88d1c756c0ea2ad7ff534dbf9c4ebd36ed73e4ca1b560533ba8ae447f75ee4","score":53.3,"scoreUnit":"percent","sourceUrl":"https://deepswe.datacurve.ai/"},{"benchmarkId":"deepswe-v1.1","evidenceKind":"official_board","harnessId":"deepswe-v1.1-reported","id":"s-deepswe-muse-spark-1.2-eb88d1c756c0","ingestRunId":"ingest-deepswe-v1.1-eb88d1c756c0","modelId":"muse-spark-1.2","n":113,"observedOn":"2026-09-22","rawPayloadHash":"eb88d1c756c0ea2ad7ff534dbf9c4ebd36ed73e4ca1b560533ba8ae447f75ee4","score":54.9,"scoreUnit":"percent","sourceUrl":"https://deepswe.datacurve.ai/"},{"benchmarkId":"deepswe-v1.1","evidenceKind":"official_board","harnessId":"deepswe-v1.1-reported","id":"s-deepswe-qwen-3.8-max-eb88d1c756c0","ingestRunId":"ingest-deepswe-v1.1-eb88d1c756c0","modelId":"qwen-3.8-max","n":113,"observedOn":"2026-09-22","rawPayloadHash":"eb88d1c756c0ea2ad7ff534dbf9c4ebd36ed73e4ca1b560533ba8ae447f75ee4","score":57.5,"scoreUnit":"percent","sourceUrl":"https://deepswe.datacurve.ai/"},{"benchmarkId":"aa-intelligence-index","evidenceKind":"official_board","harnessId":"aa-intelligence-index-official","id":"s-ds-aa","ingestRunId":"ingest-fixture-v0","modelId":"deepseek-v4-pro","n":null,"observedOn":"2026-09-02","rawPayloadHash":null,"score":53,"scoreUnit":"index","sourceUrl":"https://artificialanalysis.ai/models/comparisons/deepseek-v4-pro-vs-o3"},{"benchmarkId":"browsecomp","evidenceKind":"lab_self_report","harnessId":"browsecomp-reported","id":"s-ds-browse","ingestRunId":"ingest-fixture-v0","modelId":"deepseek-v4-pro","n":null,"observedOn":"2026-08-13","rawPayloadHash":null,"score":83.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813"},{"benchmarkId":"hle","evidenceKind":"lab_self_report","harnessId":"hle-no-tools","id":"s-ds-hle","ingestRunId":"ingest-fixture-v0","modelId":"deepseek-v4-pro","n":null,"observedOn":"2026-08-13","rawPayloadHash":null,"score":42.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813"},{"benchmarkId":"livecodebench","evidenceKind":"lab_self_report","harnessId":"livecodebench-reported","id":"s-ds-lcb","ingestRunId":"ingest-fixture-v0","modelId":"deepseek-v4-pro","n":null,"observedOn":"2026-08-13","rawPayloadHash":null,"score":93.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813"},{"benchmarkId":"swe-bench-pro","evidenceKind":"lab_self_report","harnessId":"swe-bench-pro-official","id":"s-ds-swepro","ingestRunId":"ingest-fixture-v0","modelId":"deepseek-v4-pro","n":null,"observedOn":"2026-08-13","rawPayloadHash":null,"score":55.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813"},{"benchmarkId":"swe-bench-verified","evidenceKind":"lab_self_report","harnessId":"swe-bench-verified-official","id":"s-ds-swev","ingestRunId":"ingest-fixture-v0","modelId":"deepseek-v4-pro","n":null,"observedOn":"2026-08-13","rawPayloadHash":null,"score":80.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813"},{"benchmarkId":"terminal-bench-2.1","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2.1-reported","id":"s-ds-tb21","ingestRunId":"ingest-fixture-v0","modelId":"deepseek-v4-pro","n":null,"observedOn":"2026-08-13","rawPayloadHash":null,"score":87.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813"},{"benchmarkId":"gdpval-aa","evidenceKind":"official_board","harnessId":"gdpval-aa-official","id":"s-gdpval-agnes-2.5-pro-alpha-7d1d4cbf4497","ingestRunId":"ingest-gdpval-aa-7d1d4cbf4497","modelId":"agnes-2.5-pro-alpha","n":220,"observedOn":"2026-09-19","rawPayloadHash":"7d1d4cbf44978941f5507b0a4a39e7ff28cd562698c1169505863ec7d3f401e6","score":1088,"scoreUnit":"elo","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa"},{"benchmarkId":"gdpval-aa","evidenceKind":"official_board","harnessId":"gdpval-aa-official","id":"s-gdpval-claude-opus-5-0efcc4622a07","ingestRunId":"ingest-gdpval-aa-0efcc4622a07","modelId":"claude-opus-5","n":220,"observedOn":"2026-09-30","rawPayloadHash":"0efcc4622a079db511b05929d60904491e7f3a9fe94fa2f6df2546751ac1df08","score":1708,"scoreUnit":"elo","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa"},{"benchmarkId":"gdpval-aa","evidenceKind":"official_board","harnessId":"gdpval-aa-official","id":"s-gdpval-deepseek-v4-pro-0efcc4622a07","ingestRunId":"ingest-gdpval-aa-0efcc4622a07","modelId":"deepseek-v4-pro","n":220,"observedOn":"2026-09-30","rawPayloadHash":"0efcc4622a079db511b05929d60904491e7f3a9fe94fa2f6df2546751ac1df08","score":1441,"scoreUnit":"elo","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa"},{"benchmarkId":"gdpval-aa","evidenceKind":"official_board","harnessId":"gdpval-aa-official","id":"s-gdpval-devstral-2512-0efcc4622a07","ingestRunId":"ingest-gdpval-aa-0efcc4622a07","modelId":"devstral-2512","n":220,"observedOn":"2026-09-30","rawPayloadHash":"0efcc4622a079db511b05929d60904491e7f3a9fe94fa2f6df2546751ac1df08","score":531,"scoreUnit":"elo","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa"},{"benchmarkId":"gdpval-aa","evidenceKind":"official_board","harnessId":"gdpval-aa-official","id":"s-gdpval-diffusiongemma-26b-a4b-0efcc4622a07","ingestRunId":"ingest-gdpval-aa-0efcc4622a07","modelId":"diffusiongemma-26b-a4b","n":220,"observedOn":"2026-09-30","rawPayloadHash":"0efcc4622a079db511b05929d60904491e7f3a9fe94fa2f6df2546751ac1df08","score":308,"scoreUnit":"elo","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa"},{"benchmarkId":"gdpval-aa","evidenceKind":"official_board","harnessId":"gdpval-aa-official","id":"s-gdpval-exaone-4.5-33b-0efcc4622a07","ingestRunId":"ingest-gdpval-aa-0efcc4622a07","modelId":"exaone-4.5-33b","n":220,"observedOn":"2026-09-30","rawPayloadHash":"0efcc4622a079db511b05929d60904491e7f3a9fe94fa2f6df2546751ac1df08","score":457,"scoreUnit":"elo","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa"},{"benchmarkId":"gdpval-aa","evidenceKind":"official_board","harnessId":"gdpval-aa-official","id":"s-gdpval-gemini-2.5-pro-0efcc4622a07","ingestRunId":"ingest-gdpval-aa-0efcc4622a07","modelId":"gemini-2.5-pro","n":220,"observedOn":"2026-09-30","rawPayloadHash":"0efcc4622a079db511b05929d60904491e7f3a9fe94fa2f6df2546751ac1df08","score":444,"scoreUnit":"elo","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa"},{"benchmarkId":"gdpval-aa","evidenceKind":"official_board","harnessId":"gdpval-aa-official","id":"s-gdpval-gemini-3.1-flash-lite-0efcc4622a07","ingestRunId":"ingest-gdpval-aa-0efcc4622a07","modelId":"gemini-3.1-flash-lite","n":220,"observedOn":"2026-09-30","rawPayloadHash":"0efcc4622a079db511b05929d60904491e7f3a9fe94fa2f6df2546751ac1df08","score":416,"scoreUnit":"elo","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa"},{"benchmarkId":"gdpval-aa","evidenceKind":"official_board","harnessId":"gdpval-aa-official","id":"s-gdpval-gemini-3.1-pro-0efcc4622a07","ingestRunId":"ingest-gdpval-aa-0efcc4622a07","modelId":"gemini-3.1-pro","n":220,"observedOn":"2026-09-30","rawPayloadHash":"0efcc4622a079db511b05929d60904491e7f3a9fe94fa2f6df2546751ac1df08","score":776,"scoreUnit":"elo","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa"},{"benchmarkId":"gdpval-aa","evidenceKind":"official_board","harnessId":"gdpval-aa-official","id":"s-gdpval-gemini-3.5-flash-lite-0efcc4622a07","ingestRunId":"ingest-gdpval-aa-0efcc4622a07","modelId":"gemini-3.5-flash-lite","n":220,"observedOn":"2026-09-30","rawPayloadHash":"0efcc4622a079db511b05929d60904491e7f3a9fe94fa2f6df2546751ac1df08","score":970,"scoreUnit":"elo","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa"},{"benchmarkId":"gdpval-aa","evidenceKind":"official_board","harnessId":"gdpval-aa-official","id":"s-gdpval-glm-5.3-flash-a73a52b0d734","ingestRunId":"ingest-gdpval-aa-a73a52b0d734","modelId":"glm-5.3-flash","n":220,"observedOn":"2026-09-16","rawPayloadHash":"a73a52b0d734c9de59dc6eaef4f26a6d0411a47077ac3a603b3de34880d4833e","score":1655,"scoreUnit":"elo","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa"},{"benchmarkId":"gdpval-aa","evidenceKind":"official_board","harnessId":"gdpval-aa-official","id":"s-gdpval-gpt-4-0efcc4622a07","ingestRunId":"ingest-gdpval-aa-0efcc4622a07","modelId":"gpt-4","n":220,"observedOn":"2026-09-30","rawPayloadHash":"0efcc4622a079db511b05929d60904491e7f3a9fe94fa2f6df2546751ac1df08","score":-35,"scoreUnit":"elo","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa"},{"benchmarkId":"gdpval-aa","evidenceKind":"official_board","harnessId":"gdpval-aa-official","id":"s-gdpval-gpt-4.1-mini-0efcc4622a07","ingestRunId":"ingest-gdpval-aa-0efcc4622a07","modelId":"gpt-4.1-mini","n":220,"observedOn":"2026-09-30","rawPayloadHash":"0efcc4622a079db511b05929d60904491e7f3a9fe94fa2f6df2546751ac1df08","score":256,"scoreUnit":"elo","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa"},{"benchmarkId":"gdpval-aa","evidenceKind":"official_board","harnessId":"gdpval-aa-official","id":"s-gdpval-gpt-4o-mini-0efcc4622a07","ingestRunId":"ingest-gdpval-aa-0efcc4622a07","modelId":"gpt-4o-mini","n":220,"observedOn":"2026-09-30","rawPayloadHash":"0efcc4622a079db511b05929d60904491e7f3a9fe94fa2f6df2546751ac1df08","score":-37,"scoreUnit":"elo","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa"},{"benchmarkId":"gdpval-aa","evidenceKind":"official_board","harnessId":"gdpval-aa-official","id":"s-gdpval-gpt-5.6-sol-0efcc4622a07","ingestRunId":"ingest-gdpval-aa-0efcc4622a07","modelId":"gpt-5.6-sol","n":220,"observedOn":"2026-09-30","rawPayloadHash":"0efcc4622a079db511b05929d60904491e7f3a9fe94fa2f6df2546751ac1df08","score":1588,"scoreUnit":"elo","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa"},{"benchmarkId":"gdpval-aa","evidenceKind":"official_board","harnessId":"gdpval-aa-official","id":"s-gdpval-grok-4.6-0efcc4622a07","ingestRunId":"ingest-gdpval-aa-0efcc4622a07","modelId":"grok-4.6","n":220,"observedOn":"2026-09-30","rawPayloadHash":"0efcc4622a079db511b05929d60904491e7f3a9fe94fa2f6df2546751ac1df08","score":1605,"scoreUnit":"elo","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa"},{"benchmarkId":"gdpval-aa","evidenceKind":"official_board","harnessId":"gdpval-aa-official","id":"s-gdpval-hy3-0efcc4622a07","ingestRunId":"ingest-gdpval-aa-0efcc4622a07","modelId":"hy3","n":220,"observedOn":"2026-09-30","rawPayloadHash":"0efcc4622a079db511b05929d60904491e7f3a9fe94fa2f6df2546751ac1df08","score":1045,"scoreUnit":"elo","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa"},{"benchmarkId":"gdpval-aa","evidenceKind":"official_board","harnessId":"gdpval-aa-official","id":"s-gdpval-kimi-k3-0efcc4622a07","ingestRunId":"ingest-gdpval-aa-0efcc4622a07","modelId":"kimi-k3","n":220,"observedOn":"2026-09-30","rawPayloadHash":"0efcc4622a079db511b05929d60904491e7f3a9fe94fa2f6df2546751ac1df08","score":1524,"scoreUnit":"elo","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa"},{"benchmarkId":"gdpval-aa","evidenceKind":"official_board","harnessId":"gdpval-aa-official","id":"s-gdpval-labs-devstral-small-2512-0efcc4622a07","ingestRunId":"ingest-gdpval-aa-0efcc4622a07","modelId":"labs-devstral-small-2512","n":220,"observedOn":"2026-09-30","rawPayloadHash":"0efcc4622a079db511b05929d60904491e7f3a9fe94fa2f6df2546751ac1df08","score":516,"scoreUnit":"elo","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa"},{"benchmarkId":"gdpval-aa","evidenceKind":"official_board","harnessId":"gdpval-aa-official","id":"s-gdpval-llama-4-maverick-0efcc4622a07","ingestRunId":"ingest-gdpval-aa-0efcc4622a07","modelId":"llama-4-maverick","n":220,"observedOn":"2026-09-30","rawPayloadHash":"0efcc4622a079db511b05929d60904491e7f3a9fe94fa2f6df2546751ac1df08","score":-291,"scoreUnit":"elo","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa"},{"benchmarkId":"gdpval-aa","evidenceKind":"official_board","harnessId":"gdpval-aa-official","id":"s-gdpval-magistral-medium-2509-0efcc4622a07","ingestRunId":"ingest-gdpval-aa-0efcc4622a07","modelId":"magistral-medium-2509","n":220,"observedOn":"2026-09-30","rawPayloadHash":"0efcc4622a079db511b05929d60904491e7f3a9fe94fa2f6df2546751ac1df08","score":155,"scoreUnit":"elo","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa"},{"benchmarkId":"gdpval-aa","evidenceKind":"official_board","harnessId":"gdpval-aa-official","id":"s-gdpval-magistral-small-2509-0efcc4622a07","ingestRunId":"ingest-gdpval-aa-0efcc4622a07","modelId":"magistral-small-2509","n":220,"observedOn":"2026-09-30","rawPayloadHash":"0efcc4622a079db511b05929d60904491e7f3a9fe94fa2f6df2546751ac1df08","score":-10,"scoreUnit":"elo","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa"},{"benchmarkId":"gdpval-aa","evidenceKind":"official_board","harnessId":"gdpval-aa-official","id":"s-gdpval-mimo-v2.5-0efcc4622a07","ingestRunId":"ingest-gdpval-aa-0efcc4622a07","modelId":"mimo-v2.5","n":220,"observedOn":"2026-09-30","rawPayloadHash":"0efcc4622a079db511b05929d60904491e7f3a9fe94fa2f6df2546751ac1df08","score":986,"scoreUnit":"elo","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa"},{"benchmarkId":"gdpval-aa","evidenceKind":"official_board","harnessId":"gdpval-aa-official","id":"s-gdpval-mimo-v2.5-pro-0efcc4622a07","ingestRunId":"ingest-gdpval-aa-0efcc4622a07","modelId":"mimo-v2.5-pro","n":220,"observedOn":"2026-09-30","rawPayloadHash":"0efcc4622a079db511b05929d60904491e7f3a9fe94fa2f6df2546751ac1df08","score":1107,"scoreUnit":"elo","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa"},{"benchmarkId":"gdpval-aa","evidenceKind":"official_board","harnessId":"gdpval-aa-official","id":"s-gdpval-minicpm5-2b-0efcc4622a07","ingestRunId":"ingest-gdpval-aa-0efcc4622a07","modelId":"minicpm5-2b","n":220,"observedOn":"2026-09-30","rawPayloadHash":"0efcc4622a079db511b05929d60904491e7f3a9fe94fa2f6df2546751ac1df08","score":694,"scoreUnit":"elo","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa"},{"benchmarkId":"gdpval-aa","evidenceKind":"official_board","harnessId":"gdpval-aa-official","id":"s-gdpval-minimax-m2.7-0efcc4622a07","ingestRunId":"ingest-gdpval-aa-0efcc4622a07","modelId":"minimax-m2.7","n":220,"observedOn":"2026-09-30","rawPayloadHash":"0efcc4622a079db511b05929d60904491e7f3a9fe94fa2f6df2546751ac1df08","score":999,"scoreUnit":"elo","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa"},{"benchmarkId":"gdpval-aa","evidenceKind":"official_board","harnessId":"gdpval-aa-official","id":"s-gdpval-minimax-m3-0efcc4622a07","ingestRunId":"ingest-gdpval-aa-0efcc4622a07","modelId":"minimax-m3","n":220,"observedOn":"2026-09-30","rawPayloadHash":"0efcc4622a079db511b05929d60904491e7f3a9fe94fa2f6df2546751ac1df08","score":1230,"scoreUnit":"elo","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa"},{"benchmarkId":"gdpval-aa","evidenceKind":"official_board","harnessId":"gdpval-aa-official","id":"s-gdpval-ministral-3-14b-0efcc4622a07","ingestRunId":"ingest-gdpval-aa-0efcc4622a07","modelId":"ministral-3-14b","n":220,"observedOn":"2026-09-30","rawPayloadHash":"0efcc4622a079db511b05929d60904491e7f3a9fe94fa2f6df2546751ac1df08","score":234,"scoreUnit":"elo","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa"},{"benchmarkId":"gdpval-aa","evidenceKind":"official_board","harnessId":"gdpval-aa-official","id":"s-gdpval-ministral-3-3b-0efcc4622a07","ingestRunId":"ingest-gdpval-aa-0efcc4622a07","modelId":"ministral-3-3b","n":220,"observedOn":"2026-09-30","rawPayloadHash":"0efcc4622a079db511b05929d60904491e7f3a9fe94fa2f6df2546751ac1df08","score":15,"scoreUnit":"elo","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa"},{"benchmarkId":"gdpval-aa","evidenceKind":"official_board","harnessId":"gdpval-aa-official","id":"s-gdpval-ministral-3-8b-0efcc4622a07","ingestRunId":"ingest-gdpval-aa-0efcc4622a07","modelId":"ministral-3-8b","n":220,"observedOn":"2026-09-30","rawPayloadHash":"0efcc4622a079db511b05929d60904491e7f3a9fe94fa2f6df2546751ac1df08","score":200,"scoreUnit":"elo","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa"},{"benchmarkId":"gdpval-aa","evidenceKind":"official_board","harnessId":"gdpval-aa-official","id":"s-gdpval-mistral-large-3-0efcc4622a07","ingestRunId":"ingest-gdpval-aa-0efcc4622a07","modelId":"mistral-large-3","n":220,"observedOn":"2026-09-30","rawPayloadHash":"0efcc4622a079db511b05929d60904491e7f3a9fe94fa2f6df2546751ac1df08","score":406,"scoreUnit":"elo","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa"},{"benchmarkId":"gdpval-aa","evidenceKind":"official_board","harnessId":"gdpval-aa-official","id":"s-gdpval-mistral-medium-2508-0efcc4622a07","ingestRunId":"ingest-gdpval-aa-0efcc4622a07","modelId":"mistral-medium-2508","n":220,"observedOn":"2026-09-30","rawPayloadHash":"0efcc4622a079db511b05929d60904491e7f3a9fe94fa2f6df2546751ac1df08","score":371,"scoreUnit":"elo","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa"},{"benchmarkId":"gdpval-aa","evidenceKind":"official_board","harnessId":"gdpval-aa-official","id":"s-gdpval-mistral-medium-3.5-0efcc4622a07","ingestRunId":"ingest-gdpval-aa-0efcc4622a07","modelId":"mistral-medium-3.5","n":220,"observedOn":"2026-09-30","rawPayloadHash":"0efcc4622a079db511b05929d60904491e7f3a9fe94fa2f6df2546751ac1df08","score":747,"scoreUnit":"elo","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa"},{"benchmarkId":"gdpval-aa","evidenceKind":"official_board","harnessId":"gdpval-aa-official","id":"s-gdpval-mistral-small-2506-0efcc4622a07","ingestRunId":"ingest-gdpval-aa-0efcc4622a07","modelId":"mistral-small-2506","n":220,"observedOn":"2026-09-30","rawPayloadHash":"0efcc4622a079db511b05929d60904491e7f3a9fe94fa2f6df2546751ac1df08","score":-212,"scoreUnit":"elo","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa"},{"benchmarkId":"gdpval-aa","evidenceKind":"official_board","harnessId":"gdpval-aa-official","id":"s-gdpval-nemotron-3-ultra-0efcc4622a07","ingestRunId":"ingest-gdpval-aa-0efcc4622a07","modelId":"nemotron-3-ultra","n":220,"observedOn":"2026-09-30","rawPayloadHash":"0efcc4622a079db511b05929d60904491e7f3a9fe94fa2f6df2546751ac1df08","score":1000,"scoreUnit":"elo","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa"},{"benchmarkId":"gdpval-aa","evidenceKind":"official_board","harnessId":"gdpval-aa-official","id":"s-gdpval-qwen-3.8-max-0efcc4622a07","ingestRunId":"ingest-gdpval-aa-0efcc4622a07","modelId":"qwen-3.8-max","n":220,"observedOn":"2026-09-30","rawPayloadHash":"0efcc4622a079db511b05929d60904491e7f3a9fe94fa2f6df2546751ac1df08","score":1596,"scoreUnit":"elo","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa"},{"benchmarkId":"gdpval-aa","evidenceKind":"official_board","harnessId":"gdpval-aa-official","id":"s-gdpval-qwen3.8-flash-next-0efcc4622a07","ingestRunId":"ingest-gdpval-aa-0efcc4622a07","modelId":"qwen3.8-flash-next","n":220,"observedOn":"2026-09-30","rawPayloadHash":"0efcc4622a079db511b05929d60904491e7f3a9fe94fa2f6df2546751ac1df08","score":1612,"scoreUnit":"elo","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa"},{"benchmarkId":"gdpval-aa","evidenceKind":"official_board","harnessId":"gdpval-aa-official","id":"s-gdpval-solar-open2-250b-7d1d4cbf4497","ingestRunId":"ingest-gdpval-aa-7d1d4cbf4497","modelId":"solar-open2-250b","n":220,"observedOn":"2026-09-19","rawPayloadHash":"7d1d4cbf44978941f5507b0a4a39e7ff28cd562698c1169505863ec7d3f401e6","score":1053,"scoreUnit":"elo","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa"},{"benchmarkId":"gdpval-aa","evidenceKind":"official_board","harnessId":"gdpval-aa-official","id":"s-gdpval-solar-pro3-260323-0efcc4622a07","ingestRunId":"ingest-gdpval-aa-0efcc4622a07","modelId":"solar-pro3-260323","n":220,"observedOn":"2026-09-30","rawPayloadHash":"0efcc4622a079db511b05929d60904491e7f3a9fe94fa2f6df2546751ac1df08","score":251,"scoreUnit":"elo","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa"},{"benchmarkId":"gdpval-aa","evidenceKind":"official_board","harnessId":"gdpval-aa-official","id":"s-gdpval-solar-pro4-260806-7d1d4cbf4497","ingestRunId":"ingest-gdpval-aa-7d1d4cbf4497","modelId":"solar-pro4-260806","n":220,"observedOn":"2026-09-19","rawPayloadHash":"7d1d4cbf44978941f5507b0a4a39e7ff28cd562698c1169505863ec7d3f401e6","score":1173,"scoreUnit":"elo","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa"},{"benchmarkId":"gdpval-aa","evidenceKind":"official_board","harnessId":"gdpval-aa-official","id":"s-gdpval-step-3.7-flash-0efcc4622a07","ingestRunId":"ingest-gdpval-aa-0efcc4622a07","modelId":"step-3.7-flash","n":220,"observedOn":"2026-09-30","rawPayloadHash":"0efcc4622a079db511b05929d60904491e7f3a9fe94fa2f6df2546751ac1df08","score":845,"scoreUnit":"elo","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa"},{"benchmarkId":"aa-intelligence-index","evidenceKind":"official_board","harnessId":"aa-intelligence-index-official","id":"s-gem31-aa","ingestRunId":"ingest-fixture-v0","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-06-15","rawPayloadHash":null,"score":46,"scoreUnit":"index","sourceUrl":"https://artificialanalysis.ai/articles/artificial-analysis-intelligence-index-v4-1"},{"benchmarkId":"browsecomp","evidenceKind":"lab_self_report","harnessId":"browsecomp-reported","id":"s-gem31-browse","ingestRunId":"ingest-fixture-v0","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-02-19","rawPayloadHash":null,"score":85.9,"scoreUnit":"percent","sourceUrl":"https://deepmind.google/models/model-cards/gemini-3-1-pro/"},{"benchmarkId":"swe-bench-verified","evidenceKind":"lab_self_report","harnessId":"swe-bench-verified-official","id":"s-gem31-swev","ingestRunId":"ingest-fixture-v0","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-02-19","rawPayloadHash":null,"score":80.6,"scoreUnit":"percent","sourceUrl":"https://deepmind.google/models/model-cards/gemini-3-1-pro/"},{"benchmarkId":"tau2-telecom","evidenceKind":"lab_self_report","harnessId":"tau2-telecom-reported","id":"s-gem31-tau2","ingestRunId":"ingest-fixture-v0","modelId":"gemini-3.1-pro","n":null,"observedOn":"2026-02-19","rawPayloadHash":null,"score":99.3,"scoreUnit":"percent","sourceUrl":"https://deepmind.google/models/model-cards/gemini-3-1-pro/"},{"benchmarkId":"gpqa-diamond","evidenceKind":"independent_repro","harnessId":"gpqa-diamond-reported","id":"s-gpqa-claude-fable-5-1-a84c8109bd68","ingestRunId":"ingest-gpqa-diamond-a84c8109bd68","modelId":"claude-fable-5-1","n":198,"observedOn":"2026-09-30","rawPayloadHash":"a84c8109bd682ef82be5386d44eaf3730cd9adfb325bd0eefdf9ca8a40d00018","score":93.737,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/gpqa-diamond"},{"benchmarkId":"gpqa-diamond","evidenceKind":"independent_repro","harnessId":"gpqa-diamond-reported","id":"s-gpqa-claude-fable-5-ca410b4c77b2","ingestRunId":"ingest-gpqa-diamond-ca410b4c77b2","modelId":"claude-fable-5","n":198,"observedOn":"2026-09-18","rawPayloadHash":"ca410b4c77b22441a64def7ba8a2961e8ea9ad7c4bc77d17669a8a5ba434abd7","score":92.626,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/gpqa-diamond"},{"benchmarkId":"gpqa-diamond","evidenceKind":"independent_repro","harnessId":"gpqa-diamond-reported","id":"s-gpqa-claude-opus-5-14ee8600a9d3","ingestRunId":"ingest-gpqa-diamond-14ee8600a9d3","modelId":"claude-opus-5","n":198,"observedOn":"2026-09-29","rawPayloadHash":"14ee8600a9d371c0a5471c3bfeb3d2a5eee0d2c4b90037a8a2cb9392fceadffc","score":93.232,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/gpqa-diamond"},{"benchmarkId":"gpqa-diamond","evidenceKind":"independent_repro","harnessId":"gpqa-diamond-reported","id":"s-gpqa-deepseek-v4-pro-8509afacdd47","ingestRunId":"ingest-gpqa-diamond-8509afacdd47","modelId":"deepseek-v4-pro","n":198,"observedOn":"2026-09-21","rawPayloadHash":"8509afacdd470afd77999c465efce0cb89ba00c09012d8c7ddf9640519f88a60","score":92.828,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/gpqa-diamond"},{"benchmarkId":"gpqa-diamond","evidenceKind":"independent_repro","harnessId":"gpqa-diamond-reported","id":"s-gpqa-gemini-3.1-pro-4dda8dbfc4a0","ingestRunId":"ingest-gpqa-diamond-4dda8dbfc4a0","modelId":"gemini-3.1-pro","n":198,"observedOn":"2026-09-25","rawPayloadHash":"4dda8dbfc4a017eb84ade8544e21bda55616c5f75fbe9e54db5471f0f885ff31","score":94.141,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/gpqa-diamond"},{"benchmarkId":"gpqa-diamond","evidenceKind":"independent_repro","harnessId":"gpqa-diamond-reported","id":"s-gpqa-gemini-3.5-flash-lite-a84c8109bd68","ingestRunId":"ingest-gpqa-diamond-a84c8109bd68","modelId":"gemini-3.5-flash-lite","n":198,"observedOn":"2026-09-30","rawPayloadHash":"a84c8109bd682ef82be5386d44eaf3730cd9adfb325bd0eefdf9ca8a40d00018","score":83.838,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/gpqa-diamond"},{"benchmarkId":"gpqa-diamond","evidenceKind":"independent_repro","harnessId":"gpqa-diamond-reported","id":"s-gpqa-gemini-3.7-flash-a84c8109bd68","ingestRunId":"ingest-gpqa-diamond-a84c8109bd68","modelId":"gemini-3.7-flash","n":198,"observedOn":"2026-09-30","rawPayloadHash":"a84c8109bd682ef82be5386d44eaf3730cd9adfb325bd0eefdf9ca8a40d00018","score":94.545,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/gpqa-diamond"},{"benchmarkId":"gpqa-diamond","evidenceKind":"independent_repro","harnessId":"gpqa-diamond-reported","id":"s-gpqa-gemini-3.8-flash-a84c8109bd68","ingestRunId":"ingest-gpqa-diamond-a84c8109bd68","modelId":"gemini-3.8-flash","n":198,"observedOn":"2026-09-30","rawPayloadHash":"a84c8109bd682ef82be5386d44eaf3730cd9adfb325bd0eefdf9ca8a40d00018","score":95.253,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/gpqa-diamond"},{"benchmarkId":"gpqa-diamond","evidenceKind":"independent_repro","harnessId":"gpqa-diamond-reported","id":"s-gpqa-glm-5.3-a84c8109bd68","ingestRunId":"ingest-gpqa-diamond-a84c8109bd68","modelId":"glm-5.3","n":198,"observedOn":"2026-09-30","rawPayloadHash":"a84c8109bd682ef82be5386d44eaf3730cd9adfb325bd0eefdf9ca8a40d00018","score":91.717,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/gpqa-diamond"},{"benchmarkId":"gpqa-diamond","evidenceKind":"independent_repro","harnessId":"gpqa-diamond-reported","id":"s-gpqa-glm-5.3-flash-a84c8109bd68","ingestRunId":"ingest-gpqa-diamond-a84c8109bd68","modelId":"glm-5.3-flash","n":198,"observedOn":"2026-09-30","rawPayloadHash":"a84c8109bd682ef82be5386d44eaf3730cd9adfb325bd0eefdf9ca8a40d00018","score":91.212,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/gpqa-diamond"},{"benchmarkId":"gpqa-diamond","evidenceKind":"independent_repro","harnessId":"gpqa-diamond-reported","id":"s-gpqa-gpt-5.6-luna-14ee8600a9d3","ingestRunId":"ingest-gpqa-diamond-14ee8600a9d3","modelId":"gpt-5.6-luna","n":198,"observedOn":"2026-09-29","rawPayloadHash":"14ee8600a9d371c0a5471c3bfeb3d2a5eee0d2c4b90037a8a2cb9392fceadffc","score":91.111,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/gpqa-diamond"},{"benchmarkId":"gpqa-diamond","evidenceKind":"independent_repro","harnessId":"gpqa-diamond-reported","id":"s-gpqa-gpt-5.6-sol-a84c8109bd68","ingestRunId":"ingest-gpqa-diamond-a84c8109bd68","modelId":"gpt-5.6-sol","n":198,"observedOn":"2026-09-30","rawPayloadHash":"a84c8109bd682ef82be5386d44eaf3730cd9adfb325bd0eefdf9ca8a40d00018","score":94.141,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/gpqa-diamond"},{"benchmarkId":"gpqa-diamond","evidenceKind":"independent_repro","harnessId":"gpqa-diamond-reported","id":"s-gpqa-gpt-5.6-terra-8509afacdd47","ingestRunId":"ingest-gpqa-diamond-8509afacdd47","modelId":"gpt-5.6-terra","n":198,"observedOn":"2026-09-21","rawPayloadHash":"8509afacdd470afd77999c465efce0cb89ba00c09012d8c7ddf9640519f88a60","score":92.525,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/gpqa-diamond"},{"benchmarkId":"gpqa-diamond","evidenceKind":"independent_repro","harnessId":"gpqa-diamond-reported","id":"s-gpqa-gpt-6-astra-a84c8109bd68","ingestRunId":"ingest-gpqa-diamond-a84c8109bd68","modelId":"gpt-6-astra","n":198,"observedOn":"2026-09-30","rawPayloadHash":"a84c8109bd682ef82be5386d44eaf3730cd9adfb325bd0eefdf9ca8a40d00018","score":96.061,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/gpqa-diamond"},{"benchmarkId":"gpqa-diamond","evidenceKind":"independent_repro","harnessId":"gpqa-diamond-reported","id":"s-gpqa-gpt-oss-120b-8509afacdd47","ingestRunId":"ingest-gpqa-diamond-8509afacdd47","modelId":"gpt-oss-120b","n":198,"observedOn":"2026-09-21","rawPayloadHash":"8509afacdd470afd77999c465efce0cb89ba00c09012d8c7ddf9640519f88a60","score":78.182,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/gpqa-diamond"},{"benchmarkId":"gpqa-diamond","evidenceKind":"independent_repro","harnessId":"gpqa-diamond-reported","id":"s-gpqa-grok-4.6-a84c8109bd68","ingestRunId":"ingest-gpqa-diamond-a84c8109bd68","modelId":"grok-4.6","n":198,"observedOn":"2026-09-30","rawPayloadHash":"a84c8109bd682ef82be5386d44eaf3730cd9adfb325bd0eefdf9ca8a40d00018","score":94.949,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/gpqa-diamond"},{"benchmarkId":"gpqa-diamond","evidenceKind":"independent_repro","harnessId":"gpqa-diamond-reported","id":"s-gpqa-kimi-k3-a84c8109bd68","ingestRunId":"ingest-gpqa-diamond-a84c8109bd68","modelId":"kimi-k3","n":198,"observedOn":"2026-09-30","rawPayloadHash":"a84c8109bd682ef82be5386d44eaf3730cd9adfb325bd0eefdf9ca8a40d00018","score":93.535,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/gpqa-diamond"},{"benchmarkId":"gpqa-diamond","evidenceKind":"independent_repro","harnessId":"gpqa-diamond-reported","id":"s-gpqa-minimax-m3-a84c8109bd68","ingestRunId":"ingest-gpqa-diamond-a84c8109bd68","modelId":"minimax-m3","n":198,"observedOn":"2026-09-30","rawPayloadHash":"a84c8109bd682ef82be5386d44eaf3730cd9adfb325bd0eefdf9ca8a40d00018","score":92.929,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/gpqa-diamond"},{"benchmarkId":"gpqa-diamond","evidenceKind":"independent_repro","harnessId":"gpqa-diamond-reported","id":"s-gpqa-mistral-medium-3.5-a84c8109bd68","ingestRunId":"ingest-gpqa-diamond-a84c8109bd68","modelId":"mistral-medium-3.5","n":198,"observedOn":"2026-09-30","rawPayloadHash":"a84c8109bd682ef82be5386d44eaf3730cd9adfb325bd0eefdf9ca8a40d00018","score":74.848,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/gpqa-diamond"},{"benchmarkId":"gpqa-diamond","evidenceKind":"independent_repro","harnessId":"gpqa-diamond-reported","id":"s-gpqa-muse-spark-1.3-a84c8109bd68","ingestRunId":"ingest-gpqa-diamond-a84c8109bd68","modelId":"muse-spark-1.3","n":198,"observedOn":"2026-09-30","rawPayloadHash":"a84c8109bd682ef82be5386d44eaf3730cd9adfb325bd0eefdf9ca8a40d00018","score":93.535,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/gpqa-diamond"},{"benchmarkId":"gpqa-diamond","evidenceKind":"independent_repro","harnessId":"gpqa-diamond-reported","id":"s-gpqa-nemotron-3-ultra-a84c8109bd68","ingestRunId":"ingest-gpqa-diamond-a84c8109bd68","modelId":"nemotron-3-ultra","n":198,"observedOn":"2026-09-30","rawPayloadHash":"a84c8109bd682ef82be5386d44eaf3730cd9adfb325bd0eefdf9ca8a40d00018","score":86.667,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/gpqa-diamond"},{"benchmarkId":"gpqa-diamond","evidenceKind":"independent_repro","harnessId":"gpqa-diamond-reported","id":"s-gpqa-qwen-3.8-max-a84c8109bd68","ingestRunId":"ingest-gpqa-diamond-a84c8109bd68","modelId":"qwen-3.8-max","n":198,"observedOn":"2026-09-30","rawPayloadHash":"a84c8109bd682ef82be5386d44eaf3730cd9adfb325bd0eefdf9ca8a40d00018","score":92.828,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/gpqa-diamond"},{"benchmarkId":"gpqa-diamond","evidenceKind":"independent_repro","harnessId":"gpqa-diamond-reported","id":"s-gpqa-qwen3.8-2.4t-a95b-39516434320c","ingestRunId":"ingest-gpqa-diamond-39516434320c","modelId":"qwen3.8-2.4t-a95b","n":198,"observedOn":"2026-09-17","rawPayloadHash":"39516434320c9f5f06fbbe5e0c8251ad3be5229b56e52c64236fab192a532790","score":93.535,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/gpqa-diamond"},{"benchmarkId":"gpqa-diamond","evidenceKind":"independent_repro","harnessId":"gpqa-diamond-reported","id":"s-gpqa-qwen3.8-27b-a84c8109bd68","ingestRunId":"ingest-gpqa-diamond-a84c8109bd68","modelId":"qwen3.8-27b","n":198,"observedOn":"2026-09-30","rawPayloadHash":"a84c8109bd682ef82be5386d44eaf3730cd9adfb325bd0eefdf9ca8a40d00018","score":90.505,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/evaluations/gpqa-diamond"},{"benchmarkId":"aa-intelligence-index","evidenceKind":"official_board","harnessId":"aa-intelligence-index-official","id":"s-gpt56-aa","ingestRunId":"ingest-fixture-v0","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-08-12","rawPayloadHash":null,"score":61,"scoreUnit":"index","sourceUrl":"https://artificialanalysis.ai/articles/grok-4-6-benchmarks-and-analysis"},{"benchmarkId":"browsecomp","evidenceKind":"lab_self_report","harnessId":"browsecomp-reported","id":"s-gpt56-browse","ingestRunId":"ingest-fixture-v0","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-07-09","rawPayloadHash":null,"score":90.4,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"swe-bench-pro","evidenceKind":"lab_self_report","harnessId":"swe-bench-pro-official","id":"s-gpt56-swepro","ingestRunId":"ingest-fixture-v0","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-07-09","rawPayloadHash":null,"score":64.6,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"terminal-bench-2.1","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2.1-reported","id":"s-gpt56-tb21","ingestRunId":"ingest-fixture-v0","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-07-09","rawPayloadHash":null,"score":88.8,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"terminal-bench-2.1","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2.1-codex-ultra","id":"s-gpt56-tb21-ultra","ingestRunId":"ingest-fixture-v0","modelId":"gpt-5.6-sol","n":null,"observedOn":"2026-07-09","rawPayloadHash":null,"score":91.9,"scoreUnit":"percent","sourceUrl":"https://openai.com/index/gpt-5-6/"},{"benchmarkId":"aa-intelligence-index","evidenceKind":"official_board","harnessId":"aa-intelligence-index-official","id":"s-grok46-aa","ingestRunId":"ingest-fixture-v0","modelId":"grok-4.6","n":null,"observedOn":"2026-08-12","rawPayloadHash":null,"score":61,"scoreUnit":"index","sourceUrl":"https://artificialanalysis.ai/articles/grok-4-6-benchmarks-and-analysis"},{"benchmarkId":"terminal-bench-2.1","evidenceKind":"official_board","harnessId":"terminal-bench-2.1-reported","id":"s-grok46-tb21","ingestRunId":"ingest-fixture-v0","modelId":"grok-4.6","n":null,"observedOn":"2026-08-12","rawPayloadHash":null,"score":88.4,"scoreUnit":"percent","sourceUrl":"https://artificialanalysis.ai/articles/grok-4-6-benchmarks-and-analysis"},{"benchmarkId":"hle","evidenceKind":"official_board","harnessId":"hle-no-tools","id":"s-hle-claude-opus-4-5-20251101-f549f30a8404","ingestRunId":"ingest-hle-f549f30a8404","modelId":"claude-opus-4-5-20251101","n":2500,"observedOn":"2025-11-26","rawPayloadHash":"f549f30a840454e6cd77755f3bd139f7a0e59bc238c1859c75074f78fca7d95d","score":14.16,"scoreUnit":"percent","sourceUrl":"https://scale.com/leaderboard/humanitys_last_exam"},{"benchmarkId":"hle","evidenceKind":"official_board","harnessId":"hle-no-tools","id":"s-hle-claude-opus-4-7-f549f30a8404","ingestRunId":"ingest-hle-f549f30a8404","modelId":"claude-opus-4-7","n":2500,"observedOn":"2026-04-22","rawPayloadHash":"f549f30a840454e6cd77755f3bd139f7a0e59bc238c1859c75074f78fca7d95d","score":36.2,"scoreUnit":"percent","sourceUrl":"https://scale.com/leaderboard/humanitys_last_exam"},{"benchmarkId":"hle","evidenceKind":"official_board","harnessId":"hle-no-tools","id":"s-hle-claude-sonnet-4-5-20250929-f549f30a8404","ingestRunId":"ingest-hle-f549f30a8404","modelId":"claude-sonnet-4-5-20250929","n":2500,"observedOn":"2025-10-02","rawPayloadHash":"f549f30a840454e6cd77755f3bd139f7a0e59bc238c1859c75074f78fca7d95d","score":7.52,"scoreUnit":"percent","sourceUrl":"https://scale.com/leaderboard/humanitys_last_exam"},{"benchmarkId":"hle","evidenceKind":"official_board","harnessId":"hle-no-tools","id":"s-hle-gemini-3.1-pro-f549f30a8404","ingestRunId":"ingest-hle-f549f30a8404","modelId":"gemini-3.1-pro","n":2500,"observedOn":"2026-04-10","rawPayloadHash":"f549f30a840454e6cd77755f3bd139f7a0e59bc238c1859c75074f78fca7d95d","score":46.44,"scoreUnit":"percent","sourceUrl":"https://scale.com/leaderboard/humanitys_last_exam"},{"benchmarkId":"hle","evidenceKind":"official_board","harnessId":"hle-no-tools","id":"s-hle-gemini-3.8-flash-f549f30a8404","ingestRunId":"ingest-hle-f549f30a8404","modelId":"gemini-3.8-flash","n":2500,"observedOn":"2026-09-09","rawPayloadHash":"f549f30a840454e6cd77755f3bd139f7a0e59bc238c1859c75074f78fca7d95d","score":44.52,"scoreUnit":"percent","sourceUrl":"https://scale.com/leaderboard/humanitys_last_exam"},{"benchmarkId":"hle","evidenceKind":"official_board","harnessId":"hle-no-tools","id":"s-hle-gpt-4.1-f549f30a8404","ingestRunId":"ingest-hle-f549f30a8404","modelId":"gpt-4.1","n":2500,"observedOn":"2025-04-14","rawPayloadHash":"f549f30a840454e6cd77755f3bd139f7a0e59bc238c1859c75074f78fca7d95d","score":5.4,"scoreUnit":"percent","sourceUrl":"https://scale.com/leaderboard/humanitys_last_exam"},{"benchmarkId":"hle","evidenceKind":"official_board","harnessId":"hle-no-tools","id":"s-hle-llama-4-maverick-f549f30a8404","ingestRunId":"ingest-hle-f549f30a8404","modelId":"llama-4-maverick","n":2500,"observedOn":"2025-04-10","rawPayloadHash":"f549f30a840454e6cd77755f3bd139f7a0e59bc238c1859c75074f78fca7d95d","score":5.68,"scoreUnit":"percent","sourceUrl":"https://scale.com/leaderboard/humanitys_last_exam"},{"benchmarkId":"hle","evidenceKind":"official_board","harnessId":"hle-no-tools","id":"s-hle-mistral-medium-2505-f549f30a8404","ingestRunId":"ingest-hle-f549f30a8404","modelId":"mistral-medium-2505","n":2500,"observedOn":"2025-05-13","rawPayloadHash":"f549f30a840454e6cd77755f3bd139f7a0e59bc238c1859c75074f78fca7d95d","score":4.52,"scoreUnit":"percent","sourceUrl":"https://scale.com/leaderboard/humanitys_last_exam"},{"benchmarkId":"aa-intelligence-index","evidenceKind":"official_board","harnessId":"aa-intelligence-index-official","id":"s-kimi-aa","ingestRunId":"ingest-fixture-v0","modelId":"kimi-k3","n":null,"observedOn":"2026-08-12","rawPayloadHash":null,"score":60,"scoreUnit":"index","sourceUrl":"https://artificialanalysis.ai/articles/grok-4-6-benchmarks-and-analysis"},{"benchmarkId":"browsecomp","evidenceKind":"lab_self_report","harnessId":"browsecomp-reported","id":"s-kimi-browse","ingestRunId":"ingest-fixture-v0","modelId":"kimi-k3","n":null,"observedOn":"2026-07-16","rawPayloadHash":null,"score":91.2,"scoreUnit":"percent","sourceUrl":"https://www.kimi.ai/blog/kimi-k3"},{"benchmarkId":"hle","evidenceKind":"lab_self_report","harnessId":"hle-no-tools","id":"s-kimi-hle","ingestRunId":"ingest-fixture-v0","modelId":"kimi-k3","n":null,"observedOn":"2026-07-16","rawPayloadHash":null,"score":43.5,"scoreUnit":"percent","sourceUrl":"https://www.kimi.ai/blog/kimi-k3"},{"benchmarkId":"terminal-bench-2.1","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2.1-reported","id":"s-kimi-tb21","ingestRunId":"ingest-fixture-v0","modelId":"kimi-k3","n":null,"observedOn":"2026-07-16","rawPayloadHash":null,"score":88.3,"scoreUnit":"percent","sourceUrl":"https://www.kimi.ai/blog/kimi-k3"},{"benchmarkId":"livecodebench","evidenceKind":"official_board","harnessId":"livecodebench-reported","id":"s-lcb-deepseek-r1-0528-ad3f286332a4","ingestRunId":"ingest-livecodebench-ad3f286332a4","modelId":"deepseek-r1-0528","n":454,"observedOn":"2025-05-01","rawPayloadHash":"ad3f286332a4d41c2928816e3636d084676d48b112b5250e90a8c211f0e57530","score":73.1,"scoreUnit":"percent","sourceUrl":"https://livecodebench.github.io/leaderboard.html"},{"benchmarkId":"livecodebench","evidenceKind":"official_board","harnessId":"livecodebench-reported","id":"s-lcb-exaone-4.0-32b-ad3f286332a4","ingestRunId":"ingest-livecodebench-ad3f286332a4","modelId":"exaone-4.0-32b","n":454,"observedOn":"2025-05-01","rawPayloadHash":"ad3f286332a4d41c2928816e3636d084676d48b112b5250e90a8c211f0e57530","score":70,"scoreUnit":"percent","sourceUrl":"https://livecodebench.github.io/leaderboard.html"},{"benchmarkId":"livecodebench","evidenceKind":"official_board","harnessId":"livecodebench-reported","id":"s-lcb-gpt-4-turbo-ad3f286332a4","ingestRunId":"ingest-livecodebench-ad3f286332a4","modelId":"gpt-4-turbo","n":454,"observedOn":"2025-05-01","rawPayloadHash":"ad3f286332a4d41c2928816e3636d084676d48b112b5250e90a8c211f0e57530","score":28.7,"scoreUnit":"percent","sourceUrl":"https://livecodebench.github.io/leaderboard.html"},{"benchmarkId":"livecodebench","evidenceKind":"official_board","harnessId":"livecodebench-reported","id":"s-lcb-gpt-4o-ad3f286332a4","ingestRunId":"ingest-livecodebench-ad3f286332a4","modelId":"gpt-4o","n":454,"observedOn":"2025-05-01","rawPayloadHash":"ad3f286332a4d41c2928816e3636d084676d48b112b5250e90a8c211f0e57530","score":29.5,"scoreUnit":"percent","sourceUrl":"https://livecodebench.github.io/leaderboard.html"},{"benchmarkId":"livecodebench","evidenceKind":"official_board","harnessId":"livecodebench-reported","id":"s-lcb-gpt-4o-mini-ad3f286332a4","ingestRunId":"ingest-livecodebench-ad3f286332a4","modelId":"gpt-4o-mini","n":454,"observedOn":"2025-05-01","rawPayloadHash":"ad3f286332a4d41c2928816e3636d084676d48b112b5250e90a8c211f0e57530","score":27.5,"scoreUnit":"percent","sourceUrl":"https://livecodebench.github.io/leaderboard.html"},{"benchmarkId":"livecodebench","evidenceKind":"official_board","harnessId":"livecodebench-reported","id":"s-lcb-o3-ad3f286332a4","ingestRunId":"ingest-livecodebench-ad3f286332a4","modelId":"o3","n":454,"observedOn":"2025-05-01","rawPayloadHash":"ad3f286332a4d41c2928816e3636d084676d48b112b5250e90a8c211f0e57530","score":75.8,"scoreUnit":"percent","sourceUrl":"https://livecodebench.github.io/leaderboard.html"},{"benchmarkId":"livecodebench","evidenceKind":"official_board","harnessId":"livecodebench-reported","id":"s-lcb-o3-mini-ad3f286332a4","ingestRunId":"ingest-livecodebench-ad3f286332a4","modelId":"o3-mini","n":454,"observedOn":"2025-05-01","rawPayloadHash":"ad3f286332a4d41c2928816e3636d084676d48b112b5250e90a8c211f0e57530","score":67.4,"scoreUnit":"percent","sourceUrl":"https://livecodebench.github.io/leaderboard.html"},{"benchmarkId":"livecodebench","evidenceKind":"official_board","harnessId":"livecodebench-reported","id":"s-lcb-o4-mini-ad3f286332a4","ingestRunId":"ingest-livecodebench-ad3f286332a4","modelId":"o4-mini","n":454,"observedOn":"2025-05-01","rawPayloadHash":"ad3f286332a4d41c2928816e3636d084676d48b112b5250e90a8c211f0e57530","score":80.2,"scoreUnit":"percent","sourceUrl":"https://livecodebench.github.io/leaderboard.html"},{"benchmarkId":"livecodebench","evidenceKind":"official_board","harnessId":"livecodebench-reported","id":"s-lcb-qwen3-235b-a22b-ad3f286332a4","ingestRunId":"ingest-livecodebench-ad3f286332a4","modelId":"qwen3-235b-a22b","n":454,"observedOn":"2025-05-01","rawPayloadHash":"ad3f286332a4d41c2928816e3636d084676d48b112b5250e90a8c211f0e57530","score":65.9,"scoreUnit":"percent","sourceUrl":"https://livecodebench.github.io/leaderboard.html"},{"benchmarkId":"gpqa-diamond","evidenceKind":"lab_self_report","harnessId":"gpqa-diamond-reported","id":"s-llama4-gpqa","ingestRunId":"ingest-fixture-v0","modelId":"llama-4-maverick","n":null,"observedOn":"2025-04-05","rawPayloadHash":null,"score":69.8,"scoreUnit":"percent","sourceUrl":"https://ai.meta.com/blog/llama-4-multimodal-intelligence/"},{"benchmarkId":"livecodebench","evidenceKind":"lab_self_report","harnessId":"livecodebench-reported","id":"s-llama4-lcb","ingestRunId":"ingest-fixture-v0","modelId":"llama-4-maverick","n":null,"observedOn":"2025-04-05","rawPayloadHash":null,"score":43.4,"scoreUnit":"percent","sourceUrl":"https://ai.meta.com/blog/llama-4-multimodal-intelligence/"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-c4ai-aya-expanse-32b-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"c4ai-aya-expanse-32b","n":27124,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1267.2435251794882,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-c4ai-aya-expanse-8b-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"c4ai-aya-expanse-8b","n":9818,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1223.058179234768,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-chatglm-6b-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"chatglm-6b","n":4914,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":995.7983320291153,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-chatglm2-6b-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"chatglm2-6b","n":2658,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1024.7511675185713,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-chatglm3-6b-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"chatglm3-6b","n":4658,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1056.5523744976645,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-claude-fable-5-f75684a872fa","ingestRunId":"ingest-lmarena-text-f75684a872fa","modelId":"claude-fable-5","n":30057,"observedOn":"2026-09-13","rawPayloadHash":"f75684a872fae40e83c199ec813ef2029abb1478bc4064395524a9616b9e2a28","score":1505.6827180827381,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-claude-haiku-4-5-20251001-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"claude-haiku-4-5-20251001","n":140791,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1413.9274399763758,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-claude-opus-4-5-20251101-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"claude-opus-4-5-20251101","n":72766,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1469.6786224546388,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-claude-opus-4-6-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"claude-opus-4-6","n":80836,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1497.622962014369,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-claude-opus-4-7-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"claude-opus-4-7","n":65051,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1494.6333169834716,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-claude-opus-4-8-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"claude-opus-4-8","n":62717,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1473.7466202903918,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-claude-opus-5-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"claude-opus-5","n":55063,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1491.076699495733,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-claude-sonnet-4-5-20250929-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"claude-sonnet-4-5-20250929","n":82451,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1455.1384438845143,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-claude-sonnet-4-6-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"claude-sonnet-4-6","n":70662,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1472.21847327298,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-command-a-03-2025-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"command-a-03-2025","n":55477,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1353.6392272832784,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-command-r-03-2024-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"command-r-03-2024","n":54036,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1226.8549648745675,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-command-r-08-2024-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"command-r-08-2024","n":10140,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1250.40800698115,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-command-r-plus-04-2024-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"command-r-plus-04-2024","n":77554,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1261.7540213026682,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-command-r-plus-08-2024-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"command-r-plus-08-2024","n":9866,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1276.286565745268,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-deepseek-r1-0528-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"deepseek-r1-0528","n":18065,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1421.5565824980422,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-deepseek-v3.2-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"deepseek-v3.2","n":48010,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1424.7271950130494,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-deepseek-v4-flash-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"deepseek-v4-flash","n":51871,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1436.0823976830427,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-deepseek-v4-pro-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"deepseek-v4-pro","n":57564,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1457.5191949811926,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-ernie-5.1-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"ernie-5.1","n":39423,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1467.5056411452476,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-gemini-2.5-flash-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"gemini-2.5-flash","n":125044,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1409.4164298734768,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-gemini-2.5-pro-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"gemini-2.5-pro","n":124887,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1445.5299980783273,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-gemini-3.1-pro-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"gemini-3.1-pro","n":119196,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1486.6776290065297,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-gemini-3.5-flash-lite-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"gemini-3.5-flash-lite","n":33501,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1456.0256628530637,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-gemma-1.1-2b-it-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"gemma-1.1-2b-it","n":10854,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1116.6641078163088,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-gemma-1.1-7b-it-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"gemma-1.1-7b-it","n":23893,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1182.5802074255769,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-gemma-2-27b-it-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"gemma-2-27b-it","n":75754,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1289.4810124923167,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-gemma-2-2b-it-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"gemma-2-2b-it","n":46616,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1200.0960349468587,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-gemma-2-9b-it-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"gemma-2-9b-it","n":54611,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1267.0102702858953,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-gemma-3-12b-it-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"gemma-3-12b-it","n":3829,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1341.7740362153957,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-gemma-3-27b-it-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"gemma-3-27b-it","n":46433,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1365.3688218409352,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-gemma-3-4b-it-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"gemma-3-4b-it","n":4171,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1303.2585658553041,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-gemma-3n-e4b-it-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"gemma-3n-e4b-it","n":22041,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1317.3738041296222,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-glm-4.5-air-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"glm-4.5-air","n":30453,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1373.3416091389252,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-glm-4.5-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"glm-4.5","n":23659,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1411.3301284925237,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-glm-4.5v-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"glm-4.5v","n":4809,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1352.3817079810447,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-glm-4.6-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"glm-4.6","n":35678,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1424.1924434779332,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-glm-4.6v-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"glm-4.6v","n":2830,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1378.6349322391043,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-glm-4.7-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"glm-4.7","n":12217,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1441.4392293890467,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-glm-4.7-flash-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"glm-4.7-flash","n":11983,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1365.0043242894155,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-glm-5-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"glm-5","n":29152,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1457.2041043400154,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-glm-5.1-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"glm-5.1","n":56463,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1465.0678294555669,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-glm-5.3-flash-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"glm-5.3-flash","n":19103,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1474.1694375061174,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-glm-5v-turbo-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"glm-5v-turbo","n":10393,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1433.4724251377443,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-gpt-3.5-turbo-1106-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"gpt-3.5-turbo-1106","n":16619,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1204.2168518892875,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-gpt-5.1-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"gpt-5.1","n":44329,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1438.4560034222031,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-gpt-5.2-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"gpt-5.2","n":83555,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1435.594519877053,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-gpt-5.4-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"gpt-5.4","n":67632,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1464.9635478905702,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-gpt-5.5-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"gpt-5.5","n":70635,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1476.670706708789,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-gpt-5.6-sol-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"gpt-5.6-sol","n":34258,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1483.1712763891155,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-gpt-oss-120b-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"gpt-oss-120b","n":30021,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1351.630973965965,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-gpt-oss-20b-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"gpt-oss-20b","n":10372,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1317.6052923824925,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-grok-4.3-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"grok-4.3","n":71054,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1441.59269749137,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-grok-4.5-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"grok-4.5","n":37535,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1465.413967369693,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-grok-4.6-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"grok-4.6","n":21380,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1453.3968367139787,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-hy3-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"hy3","n":9839,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1456.8157743461543,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-kimi-k2.6-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"kimi-k2.6","n":39971,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1460.9212003914142,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-kimi-k3-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"kimi-k3","n":26400,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1487.9637997404498,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-llama-3.1-70b-instruct-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"llama-3.1-70b-instruct","n":55240,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1293.3458790215661,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-llama-3.1-8b-instruct-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"llama-3.1-8b-instruct","n":49605,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1211.248999921621,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-llama-3.2-1b-instruct-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"llama-3.2-1b-instruct","n":8045,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1111.0995157476227,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-llama-3.2-3b-instruct-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"llama-3.2-3b-instruct","n":7936,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1166.7710041655016,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-llama-3.3-70b-instruct-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"llama-3.3-70b-instruct","n":54368,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1317.8378852263388,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-llama-4-maverick-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"llama-4-maverick","n":39266,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1326.9689983060455,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-llama-4-scout-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"llama-4-scout","n":29673,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1321.5646882911124,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-mimo-v2.5-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"mimo-v2.5","n":47276,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1433.5242684391908,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-mimo-v2.5-pro-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"mimo-v2.5-pro","n":69225,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1467.421865877426,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-minimax-m2-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"minimax-m2","n":6817,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1343.4191982911013,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-minimax-m2.5-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"minimax-m2.5","n":42743,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1390.8105559015992,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-minimax-m2.7-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"minimax-m2.7","n":77968,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1414.7986938471677,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-minimax-m3-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"minimax-m3","n":56646,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1440.4553191574714,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-ministral-8b-2410-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"ministral-8b-2410","n":4781,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1237.6231397763927,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-mistral-large-2407-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"mistral-large-2407","n":45459,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1314.4164104672807,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-mistral-large-2411-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"mistral-large-2411","n":28073,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1305.57381347688,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-mistral-large-3-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"mistral-large-3","n":77456,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1413.2541898108755,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-mistral-medium-2505-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"mistral-medium-2505","n":32615,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1387.4890038540175,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-mistral-medium-2508-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"mistral-medium-2508","n":95010,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1407.9272913185239,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-mistral-medium-3.5-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"mistral-medium-3.5","n":11693,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1426.1283068837586,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-mistral-small-2506-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"mistral-small-2506","n":17352,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1356.5640353433405,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-muse-spark-1.1-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"muse-spark-1.1","n":34760,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1490.7940510697133,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-nemotron-3-ultra-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"nemotron-3-ultra","n":11450,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1425.4823433303804,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-nvidia-nemotron-3-nano-30b-a3b-bf16-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"nvidia-nemotron-3-nano-30b-a3b-bf16","n":15731,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1313.429857546169,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-o3-mini-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"o3-mini","n":56572,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1348.0898084877424,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-phi-3-medium-4k-instruct-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"phi-3-medium-4k-instruct","n":25055,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1197.8874781805546,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-phi-3-mini-128k-instruct-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"phi-3-mini-128k-instruct","n":20685,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1129.8146137671774,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-phi-3-mini-4k-instruct-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"phi-3-mini-4k-instruct","n":20118,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1128.0782149968343,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-phi-3-small-8k-instruct-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"phi-3-small-8k-instruct","n":17766,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1171.0582620995779,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-phi-4-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"phi-4","n":24126,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1256.3027750745453,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-qwen-3.8-max-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"qwen-3.8-max","n":21561,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1479.3650861363328,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-qwen3-235b-a22b-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"qwen3-235b-a22b","n":25827,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1375.3330804092157,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-qwen3-235b-a22b-instruct-2507-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"qwen3-235b-a22b-instruct-2507","n":97815,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1422.3957232294892,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-qwen3-235b-a22b-thinking-2507-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"qwen3-235b-a22b-thinking-2507","n":8776,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1400.0125957354592,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-qwen3-30b-a3b-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"qwen3-30b-a3b","n":26041,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1326.5616108862557,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-qwen3-30b-a3b-instruct-2507-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"qwen3-30b-a3b-instruct-2507","n":23179,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1382.6270770634535,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-qwen3-32b-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"qwen3-32b","n":3926,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1346.954121413984,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-qwen3-coder-480b-a35b-instruct-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"qwen3-coder-480b-a35b-instruct","n":25012,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1387.384553753814,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-qwen3-next-80b-a3b-instruct-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"qwen3-next-80b-a3b-instruct","n":22602,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1399.033589914891,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-qwen3-next-80b-a3b-thinking-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"qwen3-next-80b-a3b-thinking","n":13373,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1369.3372441414779,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-qwen3-vl-235b-a22b-instruct-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"qwen3-vl-235b-a22b-instruct","n":11241,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1413.3731193122558,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-qwen3-vl-235b-a22b-thinking-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"qwen3-vl-235b-a22b-thinking","n":7797,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1394.9029719358684,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-qwen3.5-122b-a10b-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"qwen3.5-122b-a10b","n":29671,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1416.1650253363518,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-qwen3.5-27b-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"qwen3.5-27b","n":28640,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1408.679541585013,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-qwen3.5-35b-a3b-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"qwen3.5-35b-a3b","n":30420,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1394.124342656936,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-qwen3.5-397b-a17b-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"qwen3.5-397b-a17b","n":86319,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1441.7606572784205,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-qwen3.5-flash-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"qwen3.5-flash","n":61728,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1396.2661792413169,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-qwen3.6-plus-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"qwen3.6-plus","n":48095,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1443.4287870174358,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-qwen3.7-plus-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"qwen3.7-plus","n":42031,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1455.628260081713,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-qwen3.8-27b-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"qwen3.8-27b","n":16383,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1437.6461417261792,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-solar-10.7b-instruct-v1.0-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"solar-10.7b-instruct-v1.0","n":4155,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1152.4066197333007,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-solar-pro4-260806-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"solar-pro4-260806","n":13541,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1385.4970042132684,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"lmarena-text","evidenceKind":"official_board","harnessId":"lmarena-text-official","id":"s-lmarena-step-3.5-flash-c660aff40a87","ingestRunId":"ingest-lmarena-text-c660aff40a87","modelId":"step-3.5-flash","n":60243,"observedOn":"2026-09-25","rawPayloadHash":"c660aff40a87839fe137d81cd5840899c0250eeb565442388bfcd74731a0ab14","score":1393.1957613681009,"scoreUnit":"elo","sourceUrl":"https://lmarena.ai/leaderboard"},{"benchmarkId":"gpqa-diamond","evidenceKind":"independent_repro","harnessId":"gpqa-diamond-reported","id":"s-mistral-gpqa","ingestRunId":"ingest-fixture-v0","modelId":"mistral-large-3","n":null,"observedOn":"2026-09-04","rawPayloadHash":null,"score":68.43,"scoreUnit":"percent","sourceUrl":"https://vals.ai/models/mistralai_mistral-large-2512"},{"benchmarkId":"mmlu-pro","evidenceKind":"official_board","harnessId":"mmlu-pro-reported","id":"s-mmlu-pro-c4ai-aya-expanse-8b-df03094a9acf","ingestRunId":"ingest-mmlu-pro-df03094a9acf","modelId":"c4ai-aya-expanse-8b","n":12032,"observedOn":"2026-09-29","rawPayloadHash":"df03094a9acfd3c1a1bd14b2b3fdb877fbd3e53a121c77235cb5895947548703","score":33.74,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro"},{"benchmarkId":"mmlu-pro","evidenceKind":"official_board","harnessId":"mmlu-pro-reported","id":"s-mmlu-pro-deepseek-r1-0528-df03094a9acf","ingestRunId":"ingest-mmlu-pro-df03094a9acf","modelId":"deepseek-r1-0528","n":12032,"observedOn":"2026-09-29","rawPayloadHash":"df03094a9acfd3c1a1bd14b2b3fdb877fbd3e53a121c77235cb5895947548703","score":83.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro"},{"benchmarkId":"mmlu-pro","evidenceKind":"official_board","harnessId":"mmlu-pro-reported","id":"s-mmlu-pro-exaone-3.5-2.4b-instruct-df03094a9acf","ingestRunId":"ingest-mmlu-pro-df03094a9acf","modelId":"exaone-3.5-2.4b-instruct","n":12032,"observedOn":"2026-09-29","rawPayloadHash":"df03094a9acfd3c1a1bd14b2b3fdb877fbd3e53a121c77235cb5895947548703","score":39.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro"},{"benchmarkId":"mmlu-pro","evidenceKind":"official_board","harnessId":"mmlu-pro-reported","id":"s-mmlu-pro-exaone-3.5-32b-instruct-df03094a9acf","ingestRunId":"ingest-mmlu-pro-df03094a9acf","modelId":"exaone-3.5-32b-instruct","n":12032,"observedOn":"2026-09-29","rawPayloadHash":"df03094a9acfd3c1a1bd14b2b3fdb877fbd3e53a121c77235cb5895947548703","score":58.91,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro"},{"benchmarkId":"mmlu-pro","evidenceKind":"official_board","harnessId":"mmlu-pro-reported","id":"s-mmlu-pro-exaone-3.5-7.8b-instruct-df03094a9acf","ingestRunId":"ingest-mmlu-pro-df03094a9acf","modelId":"exaone-3.5-7.8b-instruct","n":12032,"observedOn":"2026-09-29","rawPayloadHash":"df03094a9acfd3c1a1bd14b2b3fdb877fbd3e53a121c77235cb5895947548703","score":46.24,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro"},{"benchmarkId":"mmlu-pro","evidenceKind":"official_board","harnessId":"mmlu-pro-reported","id":"s-mmlu-pro-gemini-2.5-pro-df03094a9acf","ingestRunId":"ingest-mmlu-pro-df03094a9acf","modelId":"gemini-2.5-pro","n":12032,"observedOn":"2026-09-29","rawPayloadHash":"df03094a9acfd3c1a1bd14b2b3fdb877fbd3e53a121c77235cb5895947548703","score":86,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro"},{"benchmarkId":"mmlu-pro","evidenceKind":"official_board","harnessId":"mmlu-pro-reported","id":"s-mmlu-pro-gemini-3.1-pro-df03094a9acf","ingestRunId":"ingest-mmlu-pro-df03094a9acf","modelId":"gemini-3.1-pro","n":12032,"observedOn":"2026-09-29","rawPayloadHash":"df03094a9acfd3c1a1bd14b2b3fdb877fbd3e53a121c77235cb5895947548703","score":91.16,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro"},{"benchmarkId":"mmlu-pro","evidenceKind":"official_board","harnessId":"mmlu-pro-reported","id":"s-mmlu-pro-gemma-2-27b-it-df03094a9acf","ingestRunId":"ingest-mmlu-pro-df03094a9acf","modelId":"gemma-2-27b-it","n":12032,"observedOn":"2026-09-29","rawPayloadHash":"df03094a9acfd3c1a1bd14b2b3fdb877fbd3e53a121c77235cb5895947548703","score":56.54,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro"},{"benchmarkId":"mmlu-pro","evidenceKind":"official_board","harnessId":"mmlu-pro-reported","id":"s-mmlu-pro-gemma-2-2b-it-df03094a9acf","ingestRunId":"ingest-mmlu-pro-df03094a9acf","modelId":"gemma-2-2b-it","n":12032,"observedOn":"2026-09-29","rawPayloadHash":"df03094a9acfd3c1a1bd14b2b3fdb877fbd3e53a121c77235cb5895947548703","score":15.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro"},{"benchmarkId":"mmlu-pro","evidenceKind":"official_board","harnessId":"mmlu-pro-reported","id":"s-mmlu-pro-gemma-2-9b-it-df03094a9acf","ingestRunId":"ingest-mmlu-pro-df03094a9acf","modelId":"gemma-2-9b-it","n":12032,"observedOn":"2026-09-29","rawPayloadHash":"df03094a9acfd3c1a1bd14b2b3fdb877fbd3e53a121c77235cb5895947548703","score":52.08,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro"},{"benchmarkId":"mmlu-pro","evidenceKind":"official_board","harnessId":"mmlu-pro-reported","id":"s-mmlu-pro-gemma-3-12b-it-df03094a9acf","ingestRunId":"ingest-mmlu-pro-df03094a9acf","modelId":"gemma-3-12b-it","n":12032,"observedOn":"2026-09-29","rawPayloadHash":"df03094a9acfd3c1a1bd14b2b3fdb877fbd3e53a121c77235cb5895947548703","score":60.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro"},{"benchmarkId":"mmlu-pro","evidenceKind":"official_board","harnessId":"mmlu-pro-reported","id":"s-mmlu-pro-gemma-3-1b-it-df03094a9acf","ingestRunId":"ingest-mmlu-pro-df03094a9acf","modelId":"gemma-3-1b-it","n":12032,"observedOn":"2026-09-29","rawPayloadHash":"df03094a9acfd3c1a1bd14b2b3fdb877fbd3e53a121c77235cb5895947548703","score":14.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro"},{"benchmarkId":"mmlu-pro","evidenceKind":"official_board","harnessId":"mmlu-pro-reported","id":"s-mmlu-pro-gemma-3-27b-it-df03094a9acf","ingestRunId":"ingest-mmlu-pro-df03094a9acf","modelId":"gemma-3-27b-it","n":12032,"observedOn":"2026-09-29","rawPayloadHash":"df03094a9acfd3c1a1bd14b2b3fdb877fbd3e53a121c77235cb5895947548703","score":67.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro"},{"benchmarkId":"mmlu-pro","evidenceKind":"official_board","harnessId":"mmlu-pro-reported","id":"s-mmlu-pro-gemma-3-4b-it-df03094a9acf","ingestRunId":"ingest-mmlu-pro-df03094a9acf","modelId":"gemma-3-4b-it","n":12032,"observedOn":"2026-09-29","rawPayloadHash":"df03094a9acfd3c1a1bd14b2b3fdb877fbd3e53a121c77235cb5895947548703","score":43.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro"},{"benchmarkId":"mmlu-pro","evidenceKind":"official_board","harnessId":"mmlu-pro-reported","id":"s-mmlu-pro-glm-4-9b-chat-df03094a9acf","ingestRunId":"ingest-mmlu-pro-df03094a9acf","modelId":"glm-4-9b-chat","n":12032,"observedOn":"2026-09-29","rawPayloadHash":"df03094a9acfd3c1a1bd14b2b3fdb877fbd3e53a121c77235cb5895947548703","score":48.01,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro"},{"benchmarkId":"mmlu-pro","evidenceKind":"official_board","harnessId":"mmlu-pro-reported","id":"s-mmlu-pro-glm-4.5-air-df03094a9acf","ingestRunId":"ingest-mmlu-pro-df03094a9acf","modelId":"glm-4.5-air","n":12032,"observedOn":"2026-09-29","rawPayloadHash":"df03094a9acfd3c1a1bd14b2b3fdb877fbd3e53a121c77235cb5895947548703","score":81.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro"},{"benchmarkId":"mmlu-pro","evidenceKind":"official_board","harnessId":"mmlu-pro-reported","id":"s-mmlu-pro-glm-4.5-df03094a9acf","ingestRunId":"ingest-mmlu-pro-df03094a9acf","modelId":"glm-4.5","n":12032,"observedOn":"2026-09-29","rawPayloadHash":"df03094a9acfd3c1a1bd14b2b3fdb877fbd3e53a121c77235cb5895947548703","score":84.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro"},{"benchmarkId":"mmlu-pro","evidenceKind":"official_board","harnessId":"mmlu-pro-reported","id":"s-mmlu-pro-glm-5-df03094a9acf","ingestRunId":"ingest-mmlu-pro-df03094a9acf","modelId":"glm-5","n":12032,"observedOn":"2026-09-29","rawPayloadHash":"df03094a9acfd3c1a1bd14b2b3fdb877fbd3e53a121c77235cb5895947548703","score":86,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro"},{"benchmarkId":"mmlu-pro","evidenceKind":"official_board","harnessId":"mmlu-pro-reported","id":"s-mmlu-pro-gpt-4-turbo-df03094a9acf","ingestRunId":"ingest-mmlu-pro-df03094a9acf","modelId":"gpt-4-turbo","n":12032,"observedOn":"2026-09-29","rawPayloadHash":"df03094a9acfd3c1a1bd14b2b3fdb877fbd3e53a121c77235cb5895947548703","score":63.71,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro"},{"benchmarkId":"mmlu-pro","evidenceKind":"official_board","harnessId":"mmlu-pro-reported","id":"s-mmlu-pro-gpt-4.1-df03094a9acf","ingestRunId":"ingest-mmlu-pro-df03094a9acf","modelId":"gpt-4.1","n":12032,"observedOn":"2026-09-29","rawPayloadHash":"df03094a9acfd3c1a1bd14b2b3fdb877fbd3e53a121c77235cb5895947548703","score":81.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro"},{"benchmarkId":"mmlu-pro","evidenceKind":"official_board","harnessId":"mmlu-pro-reported","id":"s-mmlu-pro-gpt-4o-mini-df03094a9acf","ingestRunId":"ingest-mmlu-pro-df03094a9acf","modelId":"gpt-4o-mini","n":12032,"observedOn":"2026-09-29","rawPayloadHash":"df03094a9acfd3c1a1bd14b2b3fdb877fbd3e53a121c77235cb5895947548703","score":63.09,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro"},{"benchmarkId":"mmlu-pro","evidenceKind":"official_board","harnessId":"mmlu-pro-reported","id":"s-mmlu-pro-gpt-5.1-df03094a9acf","ingestRunId":"ingest-mmlu-pro-df03094a9acf","modelId":"gpt-5.1","n":12032,"observedOn":"2026-09-29","rawPayloadHash":"df03094a9acfd3c1a1bd14b2b3fdb877fbd3e53a121c77235cb5895947548703","score":86.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro"},{"benchmarkId":"mmlu-pro","evidenceKind":"official_board","harnessId":"mmlu-pro-reported","id":"s-mmlu-pro-gpt-5.2-df03094a9acf","ingestRunId":"ingest-mmlu-pro-df03094a9acf","modelId":"gpt-5.2","n":12032,"observedOn":"2026-09-29","rawPayloadHash":"df03094a9acfd3c1a1bd14b2b3fdb877fbd3e53a121c77235cb5895947548703","score":87.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro"},{"benchmarkId":"mmlu-pro","evidenceKind":"official_board","harnessId":"mmlu-pro-reported","id":"s-mmlu-pro-gpt-5.4-df03094a9acf","ingestRunId":"ingest-mmlu-pro-df03094a9acf","modelId":"gpt-5.4","n":12032,"observedOn":"2026-09-29","rawPayloadHash":"df03094a9acfd3c1a1bd14b2b3fdb877fbd3e53a121c77235cb5895947548703","score":87.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro"},{"benchmarkId":"mmlu-pro","evidenceKind":"official_board","harnessId":"mmlu-pro-reported","id":"s-mmlu-pro-llama-3.1-405b-instruct-df03094a9acf","ingestRunId":"ingest-mmlu-pro-df03094a9acf","modelId":"llama-3.1-405b-instruct","n":12032,"observedOn":"2026-09-29","rawPayloadHash":"df03094a9acfd3c1a1bd14b2b3fdb877fbd3e53a121c77235cb5895947548703","score":73.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro"},{"benchmarkId":"mmlu-pro","evidenceKind":"official_board","harnessId":"mmlu-pro-reported","id":"s-mmlu-pro-llama-3.1-70b-instruct-df03094a9acf","ingestRunId":"ingest-mmlu-pro-df03094a9acf","modelId":"llama-3.1-70b-instruct","n":12032,"observedOn":"2026-09-29","rawPayloadHash":"df03094a9acfd3c1a1bd14b2b3fdb877fbd3e53a121c77235cb5895947548703","score":62.84,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro"},{"benchmarkId":"mmlu-pro","evidenceKind":"official_board","harnessId":"mmlu-pro-reported","id":"s-mmlu-pro-llama-3.1-8b-instruct-df03094a9acf","ingestRunId":"ingest-mmlu-pro-df03094a9acf","modelId":"llama-3.1-8b-instruct","n":12032,"observedOn":"2026-09-29","rawPayloadHash":"df03094a9acfd3c1a1bd14b2b3fdb877fbd3e53a121c77235cb5895947548703","score":44.25,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro"},{"benchmarkId":"mmlu-pro","evidenceKind":"official_board","harnessId":"mmlu-pro-reported","id":"s-mmlu-pro-llama-3.3-70b-instruct-df03094a9acf","ingestRunId":"ingest-mmlu-pro-df03094a9acf","modelId":"llama-3.3-70b-instruct","n":12032,"observedOn":"2026-09-29","rawPayloadHash":"df03094a9acfd3c1a1bd14b2b3fdb877fbd3e53a121c77235cb5895947548703","score":65.92,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro"},{"benchmarkId":"mmlu-pro","evidenceKind":"official_board","harnessId":"mmlu-pro-reported","id":"s-mmlu-pro-llama-4-maverick-df03094a9acf","ingestRunId":"ingest-mmlu-pro-df03094a9acf","modelId":"llama-4-maverick","n":12032,"observedOn":"2026-09-29","rawPayloadHash":"df03094a9acfd3c1a1bd14b2b3fdb877fbd3e53a121c77235cb5895947548703","score":80.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro"},{"benchmarkId":"mmlu-pro","evidenceKind":"official_board","harnessId":"mmlu-pro-reported","id":"s-mmlu-pro-llama-4-scout-df03094a9acf","ingestRunId":"ingest-mmlu-pro-df03094a9acf","modelId":"llama-4-scout","n":12032,"observedOn":"2026-09-29","rawPayloadHash":"df03094a9acfd3c1a1bd14b2b3fdb877fbd3e53a121c77235cb5895947548703","score":74.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro"},{"benchmarkId":"mmlu-pro","evidenceKind":"official_board","harnessId":"mmlu-pro-reported","id":"s-mmlu-pro-mimo-7b-rl-df03094a9acf","ingestRunId":"ingest-mmlu-pro-df03094a9acf","modelId":"mimo-7b-rl","n":12032,"observedOn":"2026-09-29","rawPayloadHash":"df03094a9acfd3c1a1bd14b2b3fdb877fbd3e53a121c77235cb5895947548703","score":58.6,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro"},{"benchmarkId":"mmlu-pro","evidenceKind":"official_board","harnessId":"mmlu-pro-reported","id":"s-mmlu-pro-minimax-m2-df03094a9acf","ingestRunId":"ingest-mmlu-pro-df03094a9acf","modelId":"minimax-m2","n":12032,"observedOn":"2026-09-29","rawPayloadHash":"df03094a9acfd3c1a1bd14b2b3fdb877fbd3e53a121c77235cb5895947548703","score":82,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro"},{"benchmarkId":"mmlu-pro","evidenceKind":"official_board","harnessId":"mmlu-pro-reported","id":"s-mmlu-pro-minimax-m2.1-df03094a9acf","ingestRunId":"ingest-mmlu-pro-df03094a9acf","modelId":"minimax-m2.1","n":12032,"observedOn":"2026-09-29","rawPayloadHash":"df03094a9acfd3c1a1bd14b2b3fdb877fbd3e53a121c77235cb5895947548703","score":88,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro"},{"benchmarkId":"mmlu-pro","evidenceKind":"official_board","harnessId":"mmlu-pro-reported","id":"s-mmlu-pro-minimax-m2.5-df03094a9acf","ingestRunId":"ingest-mmlu-pro-df03094a9acf","modelId":"minimax-m2.5","n":12032,"observedOn":"2026-09-29","rawPayloadHash":"df03094a9acfd3c1a1bd14b2b3fdb877fbd3e53a121c77235cb5895947548703","score":80.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro"},{"benchmarkId":"mmlu-pro","evidenceKind":"official_board","harnessId":"mmlu-pro-reported","id":"s-mmlu-pro-minimax-text-01-df03094a9acf","ingestRunId":"ingest-mmlu-pro-df03094a9acf","modelId":"minimax-text-01","n":12032,"observedOn":"2026-09-29","rawPayloadHash":"df03094a9acfd3c1a1bd14b2b3fdb877fbd3e53a121c77235cb5895947548703","score":75.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro"},{"benchmarkId":"mmlu-pro","evidenceKind":"official_board","harnessId":"mmlu-pro-reported","id":"s-mmlu-pro-phi-3.5-mini-instruct-df03094a9acf","ingestRunId":"ingest-mmlu-pro-df03094a9acf","modelId":"phi-3.5-mini-instruct","n":12032,"observedOn":"2026-09-29","rawPayloadHash":"df03094a9acfd3c1a1bd14b2b3fdb877fbd3e53a121c77235cb5895947548703","score":47.87,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro"},{"benchmarkId":"mmlu-pro","evidenceKind":"official_board","harnessId":"mmlu-pro-reported","id":"s-mmlu-pro-phi-4-df03094a9acf","ingestRunId":"ingest-mmlu-pro-df03094a9acf","modelId":"phi-4","n":12032,"observedOn":"2026-09-29","rawPayloadHash":"df03094a9acfd3c1a1bd14b2b3fdb877fbd3e53a121c77235cb5895947548703","score":70.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro"},{"benchmarkId":"mmlu-pro","evidenceKind":"official_board","harnessId":"mmlu-pro-reported","id":"s-mmlu-pro-phi-4-reasoning-df03094a9acf","ingestRunId":"ingest-mmlu-pro-df03094a9acf","modelId":"phi-4-reasoning","n":12032,"observedOn":"2026-09-29","rawPayloadHash":"df03094a9acfd3c1a1bd14b2b3fdb877fbd3e53a121c77235cb5895947548703","score":74.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro"},{"benchmarkId":"mmlu-pro","evidenceKind":"official_board","harnessId":"mmlu-pro-reported","id":"s-mmlu-pro-phi-4-reasoning-plus-df03094a9acf","ingestRunId":"ingest-mmlu-pro-df03094a9acf","modelId":"phi-4-reasoning-plus","n":12032,"observedOn":"2026-09-29","rawPayloadHash":"df03094a9acfd3c1a1bd14b2b3fdb877fbd3e53a121c77235cb5895947548703","score":76,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro"},{"benchmarkId":"mmlu-pro","evidenceKind":"official_board","harnessId":"mmlu-pro-reported","id":"s-mmlu-pro-qwen3-235b-a22b-df03094a9acf","ingestRunId":"ingest-mmlu-pro-df03094a9acf","modelId":"qwen3-235b-a22b","n":12032,"observedOn":"2026-09-29","rawPayloadHash":"df03094a9acfd3c1a1bd14b2b3fdb877fbd3e53a121c77235cb5895947548703","score":68.18,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro"},{"benchmarkId":"mmlu-pro","evidenceKind":"official_board","harnessId":"mmlu-pro-reported","id":"s-mmlu-pro-qwen3-235b-a22b-instruct-2507-df03094a9acf","ingestRunId":"ingest-mmlu-pro-df03094a9acf","modelId":"qwen3-235b-a22b-instruct-2507","n":12032,"observedOn":"2026-09-29","rawPayloadHash":"df03094a9acfd3c1a1bd14b2b3fdb877fbd3e53a121c77235cb5895947548703","score":83,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro"},{"benchmarkId":"mmlu-pro","evidenceKind":"official_board","harnessId":"mmlu-pro-reported","id":"s-mmlu-pro-qwen3-235b-a22b-thinking-2507-df03094a9acf","ingestRunId":"ingest-mmlu-pro-df03094a9acf","modelId":"qwen3-235b-a22b-thinking-2507","n":12032,"observedOn":"2026-09-29","rawPayloadHash":"df03094a9acfd3c1a1bd14b2b3fdb877fbd3e53a121c77235cb5895947548703","score":84.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro"},{"benchmarkId":"mmlu-pro","evidenceKind":"official_board","harnessId":"mmlu-pro-reported","id":"s-mmlu-pro-qwen3-30b-a3b-thinking-2507-df03094a9acf","ingestRunId":"ingest-mmlu-pro-df03094a9acf","modelId":"qwen3-30b-a3b-thinking-2507","n":12032,"observedOn":"2026-09-29","rawPayloadHash":"df03094a9acfd3c1a1bd14b2b3fdb877fbd3e53a121c77235cb5895947548703","score":80.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro"},{"benchmarkId":"mmlu-pro","evidenceKind":"official_board","harnessId":"mmlu-pro-reported","id":"s-mmlu-pro-qwen3.5-0.8b-df03094a9acf","ingestRunId":"ingest-mmlu-pro-df03094a9acf","modelId":"qwen3.5-0.8b","n":12032,"observedOn":"2026-09-29","rawPayloadHash":"df03094a9acfd3c1a1bd14b2b3fdb877fbd3e53a121c77235cb5895947548703","score":29.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro"},{"benchmarkId":"mmlu-pro","evidenceKind":"official_board","harnessId":"mmlu-pro-reported","id":"s-mmlu-pro-qwen3.5-122b-a10b-df03094a9acf","ingestRunId":"ingest-mmlu-pro-df03094a9acf","modelId":"qwen3.5-122b-a10b","n":12032,"observedOn":"2026-09-29","rawPayloadHash":"df03094a9acfd3c1a1bd14b2b3fdb877fbd3e53a121c77235cb5895947548703","score":86.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro"},{"benchmarkId":"mmlu-pro","evidenceKind":"official_board","harnessId":"mmlu-pro-reported","id":"s-mmlu-pro-qwen3.5-27b-df03094a9acf","ingestRunId":"ingest-mmlu-pro-df03094a9acf","modelId":"qwen3.5-27b","n":12032,"observedOn":"2026-09-29","rawPayloadHash":"df03094a9acfd3c1a1bd14b2b3fdb877fbd3e53a121c77235cb5895947548703","score":86.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro"},{"benchmarkId":"mmlu-pro","evidenceKind":"official_board","harnessId":"mmlu-pro-reported","id":"s-mmlu-pro-qwen3.5-2b-df03094a9acf","ingestRunId":"ingest-mmlu-pro-df03094a9acf","modelId":"qwen3.5-2b","n":12032,"observedOn":"2026-09-29","rawPayloadHash":"df03094a9acfd3c1a1bd14b2b3fdb877fbd3e53a121c77235cb5895947548703","score":55.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro"},{"benchmarkId":"mmlu-pro","evidenceKind":"official_board","harnessId":"mmlu-pro-reported","id":"s-mmlu-pro-qwen3.5-35b-a3b-df03094a9acf","ingestRunId":"ingest-mmlu-pro-df03094a9acf","modelId":"qwen3.5-35b-a3b","n":12032,"observedOn":"2026-09-29","rawPayloadHash":"df03094a9acfd3c1a1bd14b2b3fdb877fbd3e53a121c77235cb5895947548703","score":85.3,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro"},{"benchmarkId":"mmlu-pro","evidenceKind":"official_board","harnessId":"mmlu-pro-reported","id":"s-mmlu-pro-qwen3.5-397b-a17b-df03094a9acf","ingestRunId":"ingest-mmlu-pro-df03094a9acf","modelId":"qwen3.5-397b-a17b","n":12032,"observedOn":"2026-09-29","rawPayloadHash":"df03094a9acfd3c1a1bd14b2b3fdb877fbd3e53a121c77235cb5895947548703","score":87.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro"},{"benchmarkId":"mmlu-pro","evidenceKind":"official_board","harnessId":"mmlu-pro-reported","id":"s-mmlu-pro-qwen3.5-4b-df03094a9acf","ingestRunId":"ingest-mmlu-pro-df03094a9acf","modelId":"qwen3.5-4b","n":12032,"observedOn":"2026-09-29","rawPayloadHash":"df03094a9acfd3c1a1bd14b2b3fdb877fbd3e53a121c77235cb5895947548703","score":79.1,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro"},{"benchmarkId":"mmlu-pro","evidenceKind":"official_board","harnessId":"mmlu-pro-reported","id":"s-mmlu-pro-qwen3.5-9b-df03094a9acf","ingestRunId":"ingest-mmlu-pro-df03094a9acf","modelId":"qwen3.5-9b","n":12032,"observedOn":"2026-09-29","rawPayloadHash":"df03094a9acfd3c1a1bd14b2b3fdb877fbd3e53a121c77235cb5895947548703","score":82.5,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro"},{"benchmarkId":"mmlu-pro","evidenceKind":"official_board","harnessId":"mmlu-pro-reported","id":"s-mmlu-pro-seed-oss-36b-instruct-df03094a9acf","ingestRunId":"ingest-mmlu-pro-df03094a9acf","modelId":"seed-oss-36b-instruct","n":12032,"observedOn":"2026-09-29","rawPayloadHash":"df03094a9acfd3c1a1bd14b2b3fdb877fbd3e53a121c77235cb5895947548703","score":82.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro"},{"benchmarkId":"aa-intelligence-index","evidenceKind":"official_board","harnessId":"aa-intelligence-index-official","id":"s-nv-aa","ingestRunId":"ingest-fixture-v0","modelId":"nemotron-3-ultra","n":null,"observedOn":"2026-09-02","rawPayloadHash":null,"score":38,"scoreUnit":"index","sourceUrl":"https://artificialanalysis.ai/models/comparisons/nvidia-nemotron-3-ultra-550b-a55b-vs-kimi-k2-6"},{"benchmarkId":"browsecomp","evidenceKind":"lab_self_report","harnessId":"browsecomp-reported","id":"s-nv-browse","ingestRunId":"ingest-fixture-v0","modelId":"nemotron-3-ultra","n":null,"observedOn":"2026-06-04","rawPayloadHash":null,"score":44.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-NVFP4"},{"benchmarkId":"hle","evidenceKind":"lab_self_report","harnessId":"hle-no-tools","id":"s-nv-hle","ingestRunId":"ingest-fixture-v0","modelId":"nemotron-3-ultra","n":null,"observedOn":"2026-06-04","rawPayloadHash":null,"score":26.7,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-NVFP4"},{"benchmarkId":"livecodebench","evidenceKind":"lab_self_report","harnessId":"livecodebench-reported","id":"s-nv-lcb","ingestRunId":"ingest-fixture-v0","modelId":"nemotron-3-ultra","n":null,"observedOn":"2026-06-04","rawPayloadHash":null,"score":89,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-NVFP4"},{"benchmarkId":"mmlu-pro","evidenceKind":"lab_self_report","harnessId":"mmlu-pro-reported","id":"s-nv-mmlu","ingestRunId":"ingest-fixture-v0","modelId":"nemotron-3-ultra","n":null,"observedOn":"2026-06-04","rawPayloadHash":null,"score":86.8,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-NVFP4"},{"benchmarkId":"swe-bench-verified","evidenceKind":"lab_self_report","harnessId":"swe-bench-verified-official","id":"s-nv-swev","ingestRunId":"ingest-fixture-v0","modelId":"nemotron-3-ultra","n":null,"observedOn":"2026-06-04","rawPayloadHash":null,"score":71.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-NVFP4"},{"benchmarkId":"tau2-telecom","evidenceKind":"lab_self_report","harnessId":"tau2-telecom-reported","id":"s-nv-tau2","ingestRunId":"ingest-fixture-v0","modelId":"nemotron-3-ultra","n":null,"observedOn":"2026-06-04","rawPayloadHash":null,"score":92.9,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-NVFP4"},{"benchmarkId":"terminal-bench-2.1","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2.1-reported","id":"s-nv-tb21","ingestRunId":"ingest-fixture-v0","modelId":"nemotron-3-ultra","n":null,"observedOn":"2026-06-04","rawPayloadHash":null,"score":56.4,"scoreUnit":"percent","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-NVFP4"},{"benchmarkId":"aa-intelligence-index","evidenceKind":"official_board","harnessId":"aa-intelligence-index-official","id":"s-opus5-aa","ingestRunId":"ingest-fixture-v0","modelId":"claude-opus-5","n":null,"observedOn":"2026-08-12","rawPayloadHash":null,"score":63,"scoreUnit":"index","sourceUrl":"https://artificialanalysis.ai/articles/grok-4-6-benchmarks-and-analysis"},{"benchmarkId":"browsecomp","evidenceKind":"lab_self_report","harnessId":"browsecomp-reported","id":"s-opus5-browse","ingestRunId":"ingest-fixture-v0","modelId":"claude-opus-5","n":null,"observedOn":"2026-07-24","rawPayloadHash":null,"score":90.8,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/news/claude-opus-5"},{"benchmarkId":"hle","evidenceKind":"lab_self_report","harnessId":"hle-no-tools","id":"s-opus5-hle","ingestRunId":"ingest-fixture-v0","modelId":"claude-opus-5","n":null,"observedOn":"2026-07-24","rawPayloadHash":null,"score":56.3,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/news/claude-opus-5"},{"benchmarkId":"hle","evidenceKind":"lab_self_report","harnessId":"hle-with-tools","id":"s-opus5-hle-tools","ingestRunId":"ingest-fixture-v0","modelId":"claude-opus-5","n":null,"observedOn":"2026-07-24","rawPayloadHash":null,"score":64.7,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/news/claude-opus-5"},{"benchmarkId":"swe-bench-pro","evidenceKind":"lab_self_report","harnessId":"swe-bench-pro-official","id":"s-opus5-swepro","ingestRunId":"ingest-fixture-v0","modelId":"claude-opus-5","n":null,"observedOn":"2026-07-24","rawPayloadHash":null,"score":79.2,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/news/claude-opus-5"},{"benchmarkId":"swe-bench-verified","evidenceKind":"lab_self_report","harnessId":"swe-bench-verified-official","id":"s-opus5-swev","ingestRunId":"ingest-fixture-v0","modelId":"claude-opus-5","n":null,"observedOn":"2026-07-24","rawPayloadHash":null,"score":96,"scoreUnit":"percent","sourceUrl":"https://www.anthropic.com/news/claude-opus-5"},{"benchmarkId":"osworld-verified","evidenceKind":"official_board","harnessId":"osworld-verified-reported","id":"s-osworld-claude-fable-5-42c386263956","ingestRunId":"ingest-osworld-verified-42c386263956","modelId":"claude-fable-5","n":358,"observedOn":"2026-08-01","rawPayloadHash":"42c3862639569a2c686cd8665e632fd4e22ba61e0f565d97cdc01fb786f9fe1e","score":85.96,"scoreUnit":"percent","sourceUrl":"https://os-world.github.io/"},{"benchmarkId":"osworld-verified","evidenceKind":"official_board","harnessId":"osworld-verified-reported","id":"s-osworld-claude-opus-5-42c386263956","ingestRunId":"ingest-osworld-verified-42c386263956","modelId":"claude-opus-5","n":360,"observedOn":"2026-08-01","rawPayloadHash":"42c3862639569a2c686cd8665e632fd4e22ba61e0f565d97cdc01fb786f9fe1e","score":83.39,"scoreUnit":"percent","sourceUrl":"https://os-world.github.io/"},{"benchmarkId":"osworld-verified","evidenceKind":"official_board","harnessId":"osworld-verified-reported","id":"s-osworld-claude-sonnet-4-5-20250929-42c386263956","ingestRunId":"ingest-osworld-verified-42c386263956","modelId":"claude-sonnet-4-5-20250929","n":360,"observedOn":"2025-10-31","rawPayloadHash":"42c3862639569a2c686cd8665e632fd4e22ba61e0f565d97cdc01fb786f9fe1e","score":62.88,"scoreUnit":"percent","sourceUrl":"https://os-world.github.io/"},{"benchmarkId":"osworld-verified","evidenceKind":"official_board","harnessId":"osworld-verified-reported","id":"s-osworld-claude-sonnet-4-6-42c386263956","ingestRunId":"ingest-osworld-verified-42c386263956","modelId":"claude-sonnet-4-6","n":356,"observedOn":"2026-03-08","rawPayloadHash":"42c3862639569a2c686cd8665e632fd4e22ba61e0f565d97cdc01fb786f9fe1e","score":72.11,"scoreUnit":"percent","sourceUrl":"https://os-world.github.io/"},{"benchmarkId":"osworld-verified","evidenceKind":"official_board","harnessId":"osworld-verified-reported","id":"s-osworld-muse-spark-1.1-42c386263956","ingestRunId":"ingest-osworld-verified-42c386263956","modelId":"muse-spark-1.1","n":361,"observedOn":"2026-07-09","rawPayloadHash":"42c3862639569a2c686cd8665e632fd4e22ba61e0f565d97cdc01fb786f9fe1e","score":80.67,"scoreUnit":"percent","sourceUrl":"https://os-world.github.io/"},{"benchmarkId":"osworld-verified","evidenceKind":"official_board","harnessId":"osworld-verified-reported","id":"s-osworld-o3-42c386263956","ingestRunId":"ingest-osworld-verified-42c386263956","modelId":"o3","n":361,"observedOn":"2025-07-28","rawPayloadHash":"42c3862639569a2c686cd8665e632fd4e22ba61e0f565d97cdc01fb786f9fe1e","score":23,"scoreUnit":"percent","sourceUrl":"https://os-world.github.io/"},{"benchmarkId":"aa-intelligence-index","evidenceKind":"official_board","harnessId":"aa-intelligence-index-official","id":"s-qwen-aa","ingestRunId":"ingest-fixture-v0","modelId":"qwen-3.8-max","n":null,"observedOn":"2026-08-06","rawPayloadHash":null,"score":56,"scoreUnit":"index","sourceUrl":"https://x.com/ArtificialAnlys/status/2085270415614828675"},{"benchmarkId":"hle","evidenceKind":"lab_self_report","harnessId":"hle-no-tools","id":"s-qwen-hle","ingestRunId":"ingest-fixture-v0","modelId":"qwen-3.8-max","n":null,"observedOn":"2026-08-03","rawPayloadHash":null,"score":43.6,"scoreUnit":"percent","sourceUrl":"https://qwen.ai/blog?id=qwen3-max"},{"benchmarkId":"osworld-verified","evidenceKind":"lab_self_report","harnessId":"osworld-verified-reported","id":"s-qwen-osworld","ingestRunId":"ingest-fixture-v0","modelId":"qwen-3.8-max","n":null,"observedOn":"2026-08-03","rawPayloadHash":null,"score":86.1,"scoreUnit":"percent","sourceUrl":"https://qwen.ai/blog?id=qwen3-max"},{"benchmarkId":"paperbench","evidenceKind":"lab_self_report","harnessId":"paperbench-reported","id":"s-qwen-paper","ingestRunId":"ingest-fixture-v0","modelId":"qwen-3.8-max","n":null,"observedOn":"2026-08-03","rawPayloadHash":null,"score":93,"scoreUnit":"percent","sourceUrl":"https://qwen.ai/blog?id=qwen3-max"},{"benchmarkId":"swe-bench-pro","evidenceKind":"lab_self_report","harnessId":"swe-bench-pro-official","id":"s-qwen-swepro","ingestRunId":"ingest-fixture-v0","modelId":"qwen-3.8-max","n":null,"observedOn":"2026-08-03","rawPayloadHash":null,"score":67.7,"scoreUnit":"percent","sourceUrl":"https://qwen.ai/blog?id=qwen3-max"},{"benchmarkId":"terminal-bench-2.1","evidenceKind":"lab_self_report","harnessId":"terminal-bench-2.1-reported","id":"s-qwen-tb21","ingestRunId":"ingest-fixture-v0","modelId":"qwen-3.8-max","n":null,"observedOn":"2026-08-03","rawPayloadHash":null,"score":86.6,"scoreUnit":"percent","sourceUrl":"https://qwen.ai/blog?id=qwen3-max"},{"benchmarkId":"real-swe","evidenceKind":"official_board","harnessId":"real-swe-fable-5-1-claude-code","id":"s-real-swe-claude-fable-5-1-fable-5-1-claude-code-330eaaa5040c","ingestRunId":"ingest-real-swe-330eaaa5040c-fb878f2f","modelId":"claude-fable-5-1","n":10,"observedOn":"2026-09-29","rawPayloadHash":"330eaaa5040c8b0bded1fb347e077ef75cc998aedd2fec0b1b159c997a4dfae8","score":45,"scoreUnit":"percent","sourceUrl":"https://realswe.withspecific.com/"},{"benchmarkId":"real-swe","evidenceKind":"official_board","harnessId":"real-swe-gemini-3-8-flash-gemini-cli","id":"s-real-swe-gemini-3.8-flash-gemini-3-8-flash-gemini-cli-330eaaa5040c","ingestRunId":"ingest-real-swe-330eaaa5040c-fb878f2f","modelId":"gemini-3.8-flash","n":10,"observedOn":"2026-09-29","rawPayloadHash":"330eaaa5040c8b0bded1fb347e077ef75cc998aedd2fec0b1b159c997a4dfae8","score":38.75,"scoreUnit":"percent","sourceUrl":"https://realswe.withspecific.com/"},{"benchmarkId":"real-swe","evidenceKind":"official_board","harnessId":"real-swe-glm-5-3-claude-code","id":"s-real-swe-glm-5.3-glm-5-3-claude-code-330eaaa5040c","ingestRunId":"ingest-real-swe-330eaaa5040c-fb878f2f","modelId":"glm-5.3","n":10,"observedOn":"2026-09-29","rawPayloadHash":"330eaaa5040c8b0bded1fb347e077ef75cc998aedd2fec0b1b159c997a4dfae8","score":37.5,"scoreUnit":"percent","sourceUrl":"https://realswe.withspecific.com/"},{"benchmarkId":"real-swe","evidenceKind":"official_board","harnessId":"real-swe-gpt-5-6-sol-codex-cli","id":"s-real-swe-gpt-5.6-sol-gpt-5-6-sol-codex-cli-330eaaa5040c","ingestRunId":"ingest-real-swe-330eaaa5040c-fb878f2f","modelId":"gpt-5.6-sol","n":10,"observedOn":"2026-09-29","rawPayloadHash":"330eaaa5040c8b0bded1fb347e077ef75cc998aedd2fec0b1b159c997a4dfae8","score":26.25,"scoreUnit":"percent","sourceUrl":"https://realswe.withspecific.com/"},{"benchmarkId":"real-swe","evidenceKind":"official_board","harnessId":"real-swe-gpt-6-astra-codex-cli","id":"s-real-swe-gpt-6-astra-gpt-6-astra-codex-cli-330eaaa5040c","ingestRunId":"ingest-real-swe-330eaaa5040c-fb878f2f","modelId":"gpt-6-astra","n":10,"observedOn":"2026-09-29","rawPayloadHash":"330eaaa5040c8b0bded1fb347e077ef75cc998aedd2fec0b1b159c997a4dfae8","score":46.25,"scoreUnit":"percent","sourceUrl":"https://realswe.withspecific.com/"},{"benchmarkId":"real-swe","evidenceKind":"official_board","harnessId":"real-swe-grok-4-6-grok-build","id":"s-real-swe-grok-4.6-grok-4-6-grok-build-330eaaa5040c","ingestRunId":"ingest-real-swe-330eaaa5040c-fb878f2f","modelId":"grok-4.6","n":10,"observedOn":"2026-09-29","rawPayloadHash":"330eaaa5040c8b0bded1fb347e077ef75cc998aedd2fec0b1b159c997a4dfae8","score":32.5,"scoreUnit":"percent","sourceUrl":"https://realswe.withspecific.com/"},{"benchmarkId":"real-swe","evidenceKind":"official_board","harnessId":"real-swe-kimi-k3-kimi-code","id":"s-real-swe-kimi-k3-kimi-k3-kimi-code-330eaaa5040c","ingestRunId":"ingest-real-swe-330eaaa5040c-fb878f2f","modelId":"kimi-k3","n":10,"observedOn":"2026-09-29","rawPayloadHash":"330eaaa5040c8b0bded1fb347e077ef75cc998aedd2fec0b1b159c997a4dfae8","score":20,"scoreUnit":"percent","sourceUrl":"https://realswe.withspecific.com/"},{"benchmarkId":"real-swe","evidenceKind":"official_board","harnessId":"real-swe-muse-spark-1-3-muse-code","id":"s-real-swe-muse-spark-1.3-muse-spark-1-3-muse-code-330eaaa5040c","ingestRunId":"ingest-real-swe-330eaaa5040c-fb878f2f","modelId":"muse-spark-1.3","n":10,"observedOn":"2026-09-29","rawPayloadHash":"330eaaa5040c8b0bded1fb347e077ef75cc998aedd2fec0b1b159c997a4dfae8","score":36.25,"scoreUnit":"percent","sourceUrl":"https://realswe.withspecific.com/"},{"benchmarkId":"swe-bench-pro","evidenceKind":"official_board","harnessId":"swe-bench-pro-official","id":"s-swepro-claude-opus-4-6-6112d01393e8","ingestRunId":"ingest-swe-bench-pro-6112d01393e8","modelId":"claude-opus-4-6","n":731,"observedOn":"2026-04-08","rawPayloadHash":"6112d01393e8e3b9a1e9c78a6f2a4bf10d22490152d785812020f0f35ade034a","score":51.9,"scoreUnit":"percent","sourceUrl":"https://scale.com/leaderboard/swe_bench_pro_public"},{"benchmarkId":"swe-bench-pro","evidenceKind":"official_board","harnessId":"swe-bench-pro-official","id":"s-swepro-gemini-3.1-pro-6112d01393e8","ingestRunId":"ingest-swe-bench-pro-6112d01393e8","modelId":"gemini-3.1-pro","n":731,"observedOn":"2026-04-08","rawPayloadHash":"6112d01393e8e3b9a1e9c78a6f2a4bf10d22490152d785812020f0f35ade034a","score":46.1,"scoreUnit":"percent","sourceUrl":"https://scale.com/leaderboard/swe_bench_pro_public"},{"benchmarkId":"swe-bench-pro","evidenceKind":"official_board","harnessId":"swe-bench-pro-official","id":"s-swepro-gpt-5.4-6112d01393e8","ingestRunId":"ingest-swe-bench-pro-6112d01393e8","modelId":"gpt-5.4","n":731,"observedOn":"2026-04-08","rawPayloadHash":"6112d01393e8e3b9a1e9c78a6f2a4bf10d22490152d785812020f0f35ade034a","score":59.1,"scoreUnit":"percent","sourceUrl":"https://scale.com/leaderboard/swe_bench_pro_public"},{"benchmarkId":"swe-bench-pro","evidenceKind":"official_board","harnessId":"swe-bench-pro-official","id":"s-swepro-muse-spark-1.1-6112d01393e8","ingestRunId":"ingest-swe-bench-pro-6112d01393e8","modelId":"muse-spark-1.1","n":731,"observedOn":"2026-07-09","rawPayloadHash":"6112d01393e8e3b9a1e9c78a6f2a4bf10d22490152d785812020f0f35ade034a","score":61.5,"scoreUnit":"percent","sourceUrl":"https://scale.com/leaderboard/swe_bench_pro_public"},{"benchmarkId":"swe-bench-verified","evidenceKind":"official_board","harnessId":"swe-bench-verified-official","id":"s-swev-claude-haiku-4-5-20251001-077d8c0cfcca","ingestRunId":"ingest-swe-verified-077d8c0cfcca","modelId":"claude-haiku-4-5-20251001","n":500,"observedOn":"2026-02-17","rawPayloadHash":"077d8c0cfcca80400745914180b81849577184d691fd8ed35f9832d3fe5368aa","score":66.6,"scoreUnit":"percent","sourceUrl":"https://www.swebench.com/"},{"benchmarkId":"swe-bench-verified","evidenceKind":"official_board","harnessId":"swe-bench-verified-official","id":"s-swev-claude-opus-4-5-20251101-077d8c0cfcca","ingestRunId":"ingest-swe-verified-077d8c0cfcca","modelId":"claude-opus-4-5-20251101","n":500,"observedOn":"2025-11-24","rawPayloadHash":"077d8c0cfcca80400745914180b81849577184d691fd8ed35f9832d3fe5368aa","score":74.4,"scoreUnit":"percent","sourceUrl":"https://www.swebench.com/"},{"benchmarkId":"swe-bench-verified","evidenceKind":"official_board","harnessId":"swe-bench-verified-official","id":"s-swev-claude-opus-4-6-077d8c0cfcca","ingestRunId":"ingest-swe-verified-077d8c0cfcca","modelId":"claude-opus-4-6","n":500,"observedOn":"2026-02-17","rawPayloadHash":"077d8c0cfcca80400745914180b81849577184d691fd8ed35f9832d3fe5368aa","score":75.6,"scoreUnit":"percent","sourceUrl":"https://www.swebench.com/"},{"benchmarkId":"swe-bench-verified","evidenceKind":"official_board","harnessId":"swe-bench-verified-official","id":"s-swev-claude-sonnet-4-5-20250929-077d8c0cfcca","ingestRunId":"ingest-swe-verified-077d8c0cfcca","modelId":"claude-sonnet-4-5-20250929","n":500,"observedOn":"2026-02-17","rawPayloadHash":"077d8c0cfcca80400745914180b81849577184d691fd8ed35f9832d3fe5368aa","score":71.4,"scoreUnit":"percent","sourceUrl":"https://www.swebench.com/"},{"benchmarkId":"swe-bench-verified","evidenceKind":"official_board","harnessId":"swe-bench-verified-official","id":"s-swev-deepseek-v3.2-077d8c0cfcca","ingestRunId":"ingest-swe-verified-077d8c0cfcca","modelId":"deepseek-v3.2","n":500,"observedOn":"2026-02-17","rawPayloadHash":"077d8c0cfcca80400745914180b81849577184d691fd8ed35f9832d3fe5368aa","score":70,"scoreUnit":"percent","sourceUrl":"https://www.swebench.com/"},{"benchmarkId":"swe-bench-verified","evidenceKind":"official_board","harnessId":"swe-bench-verified-official","id":"s-swev-devstral-2512-077d8c0cfcca","ingestRunId":"ingest-swe-verified-077d8c0cfcca","modelId":"devstral-2512","n":500,"observedOn":"2025-12-09","rawPayloadHash":"077d8c0cfcca80400745914180b81849577184d691fd8ed35f9832d3fe5368aa","score":53.8,"scoreUnit":"percent","sourceUrl":"https://www.swebench.com/"},{"benchmarkId":"swe-bench-verified","evidenceKind":"official_board","harnessId":"swe-bench-verified-official","id":"s-swev-gemini-2.5-flash-077d8c0cfcca","ingestRunId":"ingest-swe-verified-077d8c0cfcca","modelId":"gemini-2.5-flash","n":500,"observedOn":"2025-07-26","rawPayloadHash":"077d8c0cfcca80400745914180b81849577184d691fd8ed35f9832d3fe5368aa","score":28.73,"scoreUnit":"percent","sourceUrl":"https://www.swebench.com/"},{"benchmarkId":"swe-bench-verified","evidenceKind":"official_board","harnessId":"swe-bench-verified-official","id":"s-swev-gemini-2.5-pro-077d8c0cfcca","ingestRunId":"ingest-swe-verified-077d8c0cfcca","modelId":"gemini-2.5-pro","n":500,"observedOn":"2025-07-26","rawPayloadHash":"077d8c0cfcca80400745914180b81849577184d691fd8ed35f9832d3fe5368aa","score":53.6,"scoreUnit":"percent","sourceUrl":"https://www.swebench.com/"},{"benchmarkId":"swe-bench-verified","evidenceKind":"official_board","harnessId":"swe-bench-verified-official","id":"s-swev-gemini-3-flash-preview-077d8c0cfcca","ingestRunId":"ingest-swe-verified-077d8c0cfcca","modelId":"gemini-3-flash-preview","n":500,"observedOn":"2026-02-17","rawPayloadHash":"077d8c0cfcca80400745914180b81849577184d691fd8ed35f9832d3fe5368aa","score":75.8,"scoreUnit":"percent","sourceUrl":"https://www.swebench.com/"},{"benchmarkId":"swe-bench-verified","evidenceKind":"official_board","harnessId":"swe-bench-verified-official","id":"s-swev-glm-4.5-077d8c0cfcca","ingestRunId":"ingest-swe-verified-077d8c0cfcca","modelId":"glm-4.5","n":500,"observedOn":"2025-08-22","rawPayloadHash":"077d8c0cfcca80400745914180b81849577184d691fd8ed35f9832d3fe5368aa","score":54.2,"scoreUnit":"percent","sourceUrl":"https://www.swebench.com/"},{"benchmarkId":"swe-bench-verified","evidenceKind":"official_board","harnessId":"swe-bench-verified-official","id":"s-swev-glm-4.6-077d8c0cfcca","ingestRunId":"ingest-swe-verified-077d8c0cfcca","modelId":"glm-4.6","n":500,"observedOn":"2025-12-01","rawPayloadHash":"077d8c0cfcca80400745914180b81849577184d691fd8ed35f9832d3fe5368aa","score":55.4,"scoreUnit":"percent","sourceUrl":"https://www.swebench.com/"},{"benchmarkId":"swe-bench-verified","evidenceKind":"official_board","harnessId":"swe-bench-verified-official","id":"s-swev-glm-5-077d8c0cfcca","ingestRunId":"ingest-swe-verified-077d8c0cfcca","modelId":"glm-5","n":500,"observedOn":"2026-02-17","rawPayloadHash":"077d8c0cfcca80400745914180b81849577184d691fd8ed35f9832d3fe5368aa","score":72.8,"scoreUnit":"percent","sourceUrl":"https://www.swebench.com/"},{"benchmarkId":"swe-bench-verified","evidenceKind":"official_board","harnessId":"swe-bench-verified-official","id":"s-swev-gpt-oss-120b-077d8c0cfcca","ingestRunId":"ingest-swe-verified-077d8c0cfcca","modelId":"gpt-oss-120b","n":500,"observedOn":"2025-08-07","rawPayloadHash":"077d8c0cfcca80400745914180b81849577184d691fd8ed35f9832d3fe5368aa","score":26,"scoreUnit":"percent","sourceUrl":"https://www.swebench.com/"},{"benchmarkId":"swe-bench-verified","evidenceKind":"official_board","harnessId":"swe-bench-verified-official","id":"s-swev-llama-4-maverick-077d8c0cfcca","ingestRunId":"ingest-swe-verified-077d8c0cfcca","modelId":"llama-4-maverick","n":500,"observedOn":"2025-07-20","rawPayloadHash":"077d8c0cfcca80400745914180b81849577184d691fd8ed35f9832d3fe5368aa","score":21.04,"scoreUnit":"percent","sourceUrl":"https://www.swebench.com/"},{"benchmarkId":"swe-bench-verified","evidenceKind":"official_board","harnessId":"swe-bench-verified-official","id":"s-swev-minimax-m2-077d8c0cfcca","ingestRunId":"ingest-swe-verified-077d8c0cfcca","modelId":"minimax-m2","n":500,"observedOn":"2025-11-24","rawPayloadHash":"077d8c0cfcca80400745914180b81849577184d691fd8ed35f9832d3fe5368aa","score":61,"scoreUnit":"percent","sourceUrl":"https://www.swebench.com/"},{"benchmarkId":"swe-bench-verified","evidenceKind":"official_board","harnessId":"swe-bench-verified-official","id":"s-swev-minimax-m2.5-077d8c0cfcca","ingestRunId":"ingest-swe-verified-077d8c0cfcca","modelId":"minimax-m2.5","n":500,"observedOn":"2026-02-17","rawPayloadHash":"077d8c0cfcca80400745914180b81849577184d691fd8ed35f9832d3fe5368aa","score":75.8,"scoreUnit":"percent","sourceUrl":"https://www.swebench.com/"},{"benchmarkId":"swe-bench-verified","evidenceKind":"official_board","harnessId":"swe-bench-verified-official","id":"s-swev-o3-077d8c0cfcca","ingestRunId":"ingest-swe-verified-077d8c0cfcca","modelId":"o3","n":500,"observedOn":"2025-07-26","rawPayloadHash":"077d8c0cfcca80400745914180b81849577184d691fd8ed35f9832d3fe5368aa","score":58.4,"scoreUnit":"percent","sourceUrl":"https://www.swebench.com/"},{"benchmarkId":"swe-bench-verified","evidenceKind":"official_board","harnessId":"swe-bench-verified-official","id":"s-swev-o4-mini-077d8c0cfcca","ingestRunId":"ingest-swe-verified-077d8c0cfcca","modelId":"o4-mini","n":500,"observedOn":"2025-07-26","rawPayloadHash":"077d8c0cfcca80400745914180b81849577184d691fd8ed35f9832d3fe5368aa","score":45,"scoreUnit":"percent","sourceUrl":"https://www.swebench.com/"},{"benchmarkId":"swe-bench-verified","evidenceKind":"official_board","harnessId":"swe-bench-verified-official","id":"s-swev-qwen3-coder-480b-a35b-instruct-077d8c0cfcca","ingestRunId":"ingest-swe-verified-077d8c0cfcca","modelId":"qwen3-coder-480b-a35b-instruct","n":500,"observedOn":"2025-08-02","rawPayloadHash":"077d8c0cfcca80400745914180b81849577184d691fd8ed35f9832d3fe5368aa","score":55.4,"scoreUnit":"percent","sourceUrl":"https://www.swebench.com/"},{"benchmarkId":"terminal-bench-2.1","evidenceKind":"official_board","harnessId":"terminal-bench-2.1-reported","id":"s-tb21-gemini-3.1-pro-a98710a64150","ingestRunId":"ingest-terminal-bench-2.1-a98710a64150","modelId":"gemini-3.1-pro","n":445,"observedOn":"2026-02-19","rawPayloadHash":"a98710a64150db56e74858137de32081bc1e20692a2d140544903b525e654d41","score":65.62,"scoreUnit":"percent","sourceUrl":"https://www.tbench.ai/?version=2.1"},{"benchmarkId":"terminal-bench-2.1","evidenceKind":"official_board","harnessId":"terminal-bench-2.1-reported","id":"s-tb21-gpt-5.5-a98710a64150","ingestRunId":"ingest-terminal-bench-2.1-a98710a64150","modelId":"gpt-5.5","n":445,"observedOn":"2026-04-23","rawPayloadHash":"a98710a64150db56e74858137de32081bc1e20692a2d140544903b525e654d41","score":77.98,"scoreUnit":"percent","sourceUrl":"https://www.tbench.ai/?version=2.1"},{"benchmarkId":"terminal-bench-4.0","evidenceKind":"official_board","harnessId":"terminal-bench-4.0-reported","id":"s-tb40-claude-fable-5-1-8f0c0dd2a57b","ingestRunId":"ingest-terminal-bench-4.0-8f0c0dd2a57b","modelId":"claude-fable-5-1","n":330,"observedOn":"2026-09-01","rawPayloadHash":"8f0c0dd2a57be01a51043c0bbeefdadb7758503bd4df7a44830d79e17ca20074","score":57.88,"scoreUnit":"percent","sourceUrl":"https://www.tbench.ai/"},{"benchmarkId":"terminal-bench-4.0","evidenceKind":"official_board","harnessId":"terminal-bench-4.0-reported","id":"s-tb40-claude-fable-5-8f0c0dd2a57b","ingestRunId":"ingest-terminal-bench-4.0-8f0c0dd2a57b","modelId":"claude-fable-5","n":330,"observedOn":"2026-06-09","rawPayloadHash":"8f0c0dd2a57be01a51043c0bbeefdadb7758503bd4df7a44830d79e17ca20074","score":44.55,"scoreUnit":"percent","sourceUrl":"https://www.tbench.ai/"},{"benchmarkId":"terminal-bench-4.0","evidenceKind":"official_board","harnessId":"terminal-bench-4.0-reported","id":"s-tb40-claude-opus-4-8-8f0c0dd2a57b","ingestRunId":"ingest-terminal-bench-4.0-8f0c0dd2a57b","modelId":"claude-opus-4-8","n":330,"observedOn":"2026-05-28","rawPayloadHash":"8f0c0dd2a57be01a51043c0bbeefdadb7758503bd4df7a44830d79e17ca20074","score":23.64,"scoreUnit":"percent","sourceUrl":"https://www.tbench.ai/"},{"benchmarkId":"terminal-bench-4.0","evidenceKind":"official_board","harnessId":"terminal-bench-4.0-reported","id":"s-tb40-claude-opus-5-8f0c0dd2a57b","ingestRunId":"ingest-terminal-bench-4.0-8f0c0dd2a57b","modelId":"claude-opus-5","n":330,"observedOn":"2026-07-24","rawPayloadHash":"8f0c0dd2a57be01a51043c0bbeefdadb7758503bd4df7a44830d79e17ca20074","score":53.94,"scoreUnit":"percent","sourceUrl":"https://www.tbench.ai/"},{"benchmarkId":"terminal-bench-4.0","evidenceKind":"official_board","harnessId":"terminal-bench-4.0-reported","id":"s-tb40-claude-sonnet-5-8f0c0dd2a57b","ingestRunId":"ingest-terminal-bench-4.0-8f0c0dd2a57b","modelId":"claude-sonnet-5","n":330,"observedOn":"2026-06-30","rawPayloadHash":"8f0c0dd2a57be01a51043c0bbeefdadb7758503bd4df7a44830d79e17ca20074","score":12.42,"scoreUnit":"percent","sourceUrl":"https://www.tbench.ai/"},{"benchmarkId":"terminal-bench-4.0","evidenceKind":"official_board","harnessId":"terminal-bench-4.0-reported","id":"s-tb40-gemini-3.7-flash-8f0c0dd2a57b","ingestRunId":"ingest-terminal-bench-4.0-8f0c0dd2a57b","modelId":"gemini-3.7-flash","n":330,"observedOn":"2026-08-13","rawPayloadHash":"8f0c0dd2a57be01a51043c0bbeefdadb7758503bd4df7a44830d79e17ca20074","score":11.21,"scoreUnit":"percent","sourceUrl":"https://www.tbench.ai/"},{"benchmarkId":"terminal-bench-4.0","evidenceKind":"official_board","harnessId":"terminal-bench-4.0-reported","id":"s-tb40-gemini-3.8-flash-8f0c0dd2a57b","ingestRunId":"ingest-terminal-bench-4.0-8f0c0dd2a57b","modelId":"gemini-3.8-flash","n":330,"observedOn":"2026-09-02","rawPayloadHash":"8f0c0dd2a57be01a51043c0bbeefdadb7758503bd4df7a44830d79e17ca20074","score":19.09,"scoreUnit":"percent","sourceUrl":"https://www.tbench.ai/"},{"benchmarkId":"terminal-bench-4.0","evidenceKind":"official_board","harnessId":"terminal-bench-4.0-reported","id":"s-tb40-glm-5.3-8f0c0dd2a57b","ingestRunId":"ingest-terminal-bench-4.0-8f0c0dd2a57b","modelId":"glm-5.3","n":330,"observedOn":"2026-08-14","rawPayloadHash":"8f0c0dd2a57be01a51043c0bbeefdadb7758503bd4df7a44830d79e17ca20074","score":41.82,"scoreUnit":"percent","sourceUrl":"https://www.tbench.ai/"},{"benchmarkId":"terminal-bench-4.0","evidenceKind":"official_board","harnessId":"terminal-bench-4.0-reported","id":"s-tb40-gpt-5.6-luna-8f0c0dd2a57b","ingestRunId":"ingest-terminal-bench-4.0-8f0c0dd2a57b","modelId":"gpt-5.6-luna","n":330,"observedOn":"2026-06-26","rawPayloadHash":"8f0c0dd2a57be01a51043c0bbeefdadb7758503bd4df7a44830d79e17ca20074","score":17.27,"scoreUnit":"percent","sourceUrl":"https://www.tbench.ai/"},{"benchmarkId":"terminal-bench-4.0","evidenceKind":"official_board","harnessId":"terminal-bench-4.0-reported","id":"s-tb40-gpt-5.6-sol-8f0c0dd2a57b","ingestRunId":"ingest-terminal-bench-4.0-8f0c0dd2a57b","modelId":"gpt-5.6-sol","n":330,"observedOn":"2026-06-26","rawPayloadHash":"8f0c0dd2a57be01a51043c0bbeefdadb7758503bd4df7a44830d79e17ca20074","score":37.27,"scoreUnit":"percent","sourceUrl":"https://www.tbench.ai/"},{"benchmarkId":"terminal-bench-4.0","evidenceKind":"official_board","harnessId":"terminal-bench-4.0-reported","id":"s-tb40-gpt-5.6-terra-8f0c0dd2a57b","ingestRunId":"ingest-terminal-bench-4.0-8f0c0dd2a57b","modelId":"gpt-5.6-terra","n":330,"observedOn":"2026-06-26","rawPayloadHash":"8f0c0dd2a57be01a51043c0bbeefdadb7758503bd4df7a44830d79e17ca20074","score":21.52,"scoreUnit":"percent","sourceUrl":"https://www.tbench.ai/"},{"benchmarkId":"terminal-bench-4.0","evidenceKind":"official_board","harnessId":"terminal-bench-4.0-reported","id":"s-tb40-gpt-6-astra-8f0c0dd2a57b","ingestRunId":"ingest-terminal-bench-4.0-8f0c0dd2a57b","modelId":"gpt-6-astra","n":330,"observedOn":"2026-09-03","rawPayloadHash":"8f0c0dd2a57be01a51043c0bbeefdadb7758503bd4df7a44830d79e17ca20074","score":58.18,"scoreUnit":"percent","sourceUrl":"https://www.tbench.ai/"},{"benchmarkId":"terminal-bench-4.0","evidenceKind":"official_board","harnessId":"terminal-bench-4.0-reported","id":"s-tb40-grok-4.5-8f0c0dd2a57b","ingestRunId":"ingest-terminal-bench-4.0-8f0c0dd2a57b","modelId":"grok-4.5","n":330,"observedOn":"2026-07-16","rawPayloadHash":"8f0c0dd2a57be01a51043c0bbeefdadb7758503bd4df7a44830d79e17ca20074","score":12.42,"scoreUnit":"percent","sourceUrl":"https://www.tbench.ai/"},{"benchmarkId":"terminal-bench-4.0","evidenceKind":"official_board","harnessId":"terminal-bench-4.0-reported","id":"s-tb40-grok-4.6-8f0c0dd2a57b","ingestRunId":"ingest-terminal-bench-4.0-8f0c0dd2a57b","modelId":"grok-4.6","n":330,"observedOn":"2026-08-12","rawPayloadHash":"8f0c0dd2a57be01a51043c0bbeefdadb7758503bd4df7a44830d79e17ca20074","score":20.3,"scoreUnit":"percent","sourceUrl":"https://www.tbench.ai/"},{"benchmarkId":"terminal-bench-4.0","evidenceKind":"official_board","harnessId":"terminal-bench-4.0-reported","id":"s-tb40-grok-4.7-8f0c0dd2a57b","ingestRunId":"ingest-terminal-bench-4.0-8f0c0dd2a57b","modelId":"grok-4.7","n":330,"observedOn":"2026-09-21","rawPayloadHash":"8f0c0dd2a57be01a51043c0bbeefdadb7758503bd4df7a44830d79e17ca20074","score":37.58,"scoreUnit":"percent","sourceUrl":"https://www.tbench.ai/"},{"benchmarkId":"vulcanbench-v3","evidenceKind":"official_board","harnessId":"vulcanbench-v3-10-opus5-effort-high","id":"s-vulcanbench-claude-opus-5-10-opus5-effort-high-da2f9d24acc8","ingestRunId":"ingest-vulcanbench-v3-da2f9d24acc8-fb878f2f","modelId":"claude-opus-5","n":23,"observedOn":"2026-07-26","rawPayloadHash":"da2f9d24acc827209e42b26bace6a781f543cb219ff7eef480cfd074db2fac72","score":78.3,"scoreUnit":"percent","sourceUrl":"https://vulcanbench.com/benchmarks/10-opus5-effort.html"},{"benchmarkId":"vulcanbench-v3","evidenceKind":"official_board","harnessId":"vulcanbench-v3-10-opus5-effort-low","id":"s-vulcanbench-claude-opus-5-10-opus5-effort-low-da2f9d24acc8","ingestRunId":"ingest-vulcanbench-v3-da2f9d24acc8-fb878f2f","modelId":"claude-opus-5","n":23,"observedOn":"2026-07-26","rawPayloadHash":"da2f9d24acc827209e42b26bace6a781f543cb219ff7eef480cfd074db2fac72","score":87,"scoreUnit":"percent","sourceUrl":"https://vulcanbench.com/benchmarks/10-opus5-effort.html"},{"benchmarkId":"vulcanbench-v3","evidenceKind":"official_board","harnessId":"vulcanbench-v3-10-opus5-effort-medium","id":"s-vulcanbench-claude-opus-5-10-opus5-effort-medium-da2f9d24acc8","ingestRunId":"ingest-vulcanbench-v3-da2f9d24acc8-fb878f2f","modelId":"claude-opus-5","n":23,"observedOn":"2026-07-26","rawPayloadHash":"da2f9d24acc827209e42b26bace6a781f543cb219ff7eef480cfd074db2fac72","score":82.6,"scoreUnit":"percent","sourceUrl":"https://vulcanbench.com/benchmarks/10-opus5-effort.html"},{"benchmarkId":"vulcanbench-v3","evidenceKind":"official_board","harnessId":"vulcanbench-v3-18-glm53-zcode-harness-high","id":"s-vulcanbench-glm-5.3-18-glm53-zcode-harness-high-da2f9d24acc8","ingestRunId":"ingest-vulcanbench-v3-da2f9d24acc8-fb878f2f","modelId":"glm-5.3","n":23,"observedOn":"2026-08-24","rawPayloadHash":"da2f9d24acc827209e42b26bace6a781f543cb219ff7eef480cfd074db2fac72","score":73.9,"scoreUnit":"percent","sourceUrl":"https://vulcanbench.com/benchmarks/18-glm53-zcode-harness.html"},{"benchmarkId":"vulcanbench-v3","evidenceKind":"official_board","harnessId":"vulcanbench-v3-18-glm53-zcode-harness-low","id":"s-vulcanbench-glm-5.3-18-glm53-zcode-harness-low-da2f9d24acc8","ingestRunId":"ingest-vulcanbench-v3-da2f9d24acc8-fb878f2f","modelId":"glm-5.3","n":23,"observedOn":"2026-08-24","rawPayloadHash":"da2f9d24acc827209e42b26bace6a781f543cb219ff7eef480cfd074db2fac72","score":78.3,"scoreUnit":"percent","sourceUrl":"https://vulcanbench.com/benchmarks/18-glm53-zcode-harness.html"},{"benchmarkId":"vulcanbench-v3","evidenceKind":"official_board","harnessId":"vulcanbench-v3-18-glm53-zcode-harness-max","id":"s-vulcanbench-glm-5.3-18-glm53-zcode-harness-max-da2f9d24acc8","ingestRunId":"ingest-vulcanbench-v3-da2f9d24acc8-fb878f2f","modelId":"glm-5.3","n":23,"observedOn":"2026-08-24","rawPayloadHash":"da2f9d24acc827209e42b26bace6a781f543cb219ff7eef480cfd074db2fac72","score":65.2,"scoreUnit":"percent","sourceUrl":"https://vulcanbench.com/benchmarks/18-glm53-zcode-harness.html"},{"benchmarkId":"vulcanbench-v3","evidenceKind":"official_board","harnessId":"vulcanbench-v3-07-grok-fable-sol-high","id":"s-vulcanbench-gpt-5.6-sol-07-grok-fable-sol-high-da2f9d24acc8","ingestRunId":"ingest-vulcanbench-v3-da2f9d24acc8-fb878f2f","modelId":"gpt-5.6-sol","n":23,"observedOn":"2026-07-12","rawPayloadHash":"da2f9d24acc827209e42b26bace6a781f543cb219ff7eef480cfd074db2fac72","score":87,"scoreUnit":"percent","sourceUrl":"https://vulcanbench.com/benchmarks/07-grok-fable-sol.html"},{"benchmarkId":"vulcanbench-v3","evidenceKind":"official_board","harnessId":"vulcanbench-v3-07-grok-fable-sol-low","id":"s-vulcanbench-gpt-5.6-sol-07-grok-fable-sol-low-da2f9d24acc8","ingestRunId":"ingest-vulcanbench-v3-da2f9d24acc8-fb878f2f","modelId":"gpt-5.6-sol","n":23,"observedOn":"2026-07-12","rawPayloadHash":"da2f9d24acc827209e42b26bace6a781f543cb219ff7eef480cfd074db2fac72","score":78.3,"scoreUnit":"percent","sourceUrl":"https://vulcanbench.com/benchmarks/07-grok-fable-sol.html"},{"benchmarkId":"vulcanbench-v3","evidenceKind":"official_board","harnessId":"vulcanbench-v3-07-grok-fable-sol-medium","id":"s-vulcanbench-gpt-5.6-sol-07-grok-fable-sol-medium-da2f9d24acc8","ingestRunId":"ingest-vulcanbench-v3-da2f9d24acc8-fb878f2f","modelId":"gpt-5.6-sol","n":23,"observedOn":"2026-07-12","rawPayloadHash":"da2f9d24acc827209e42b26bace6a781f543cb219ff7eef480cfd074db2fac72","score":82.6,"scoreUnit":"percent","sourceUrl":"https://vulcanbench.com/benchmarks/07-grok-fable-sol.html"},{"benchmarkId":"vulcanbench-v3","evidenceKind":"official_board","harnessId":"vulcanbench-v3-07-grok-fable-sol-high","id":"s-vulcanbench-grok-4.5-07-grok-fable-sol-high-da2f9d24acc8","ingestRunId":"ingest-vulcanbench-v3-da2f9d24acc8-fb878f2f","modelId":"grok-4.5","n":23,"observedOn":"2026-07-12","rawPayloadHash":"da2f9d24acc827209e42b26bace6a781f543cb219ff7eef480cfd074db2fac72","score":91.3,"scoreUnit":"percent","sourceUrl":"https://vulcanbench.com/benchmarks/07-grok-fable-sol.html"},{"benchmarkId":"vulcanbench-v3","evidenceKind":"official_board","harnessId":"vulcanbench-v3-07-grok-fable-sol-low","id":"s-vulcanbench-grok-4.5-07-grok-fable-sol-low-da2f9d24acc8","ingestRunId":"ingest-vulcanbench-v3-da2f9d24acc8-fb878f2f","modelId":"grok-4.5","n":23,"observedOn":"2026-07-12","rawPayloadHash":"da2f9d24acc827209e42b26bace6a781f543cb219ff7eef480cfd074db2fac72","score":82.6,"scoreUnit":"percent","sourceUrl":"https://vulcanbench.com/benchmarks/07-grok-fable-sol.html"},{"benchmarkId":"vulcanbench-v3","evidenceKind":"official_board","harnessId":"vulcanbench-v3-07-grok-fable-sol-medium","id":"s-vulcanbench-grok-4.5-07-grok-fable-sol-medium-da2f9d24acc8","ingestRunId":"ingest-vulcanbench-v3-da2f9d24acc8-fb878f2f","modelId":"grok-4.5","n":23,"observedOn":"2026-07-12","rawPayloadHash":"da2f9d24acc827209e42b26bace6a781f543cb219ff7eef480cfd074db2fac72","score":91.3,"scoreUnit":"percent","sourceUrl":"https://vulcanbench.com/benchmarks/07-grok-fable-sol.html"},{"benchmarkId":"vulcanbench-v3","evidenceKind":"official_board","harnessId":"vulcanbench-v3-14-grok-46-effort-high","id":"s-vulcanbench-grok-4.6-14-grok-46-effort-high-da2f9d24acc8","ingestRunId":"ingest-vulcanbench-v3-da2f9d24acc8-fb878f2f","modelId":"grok-4.6","n":23,"observedOn":"2026-08-12","rawPayloadHash":"da2f9d24acc827209e42b26bace6a781f543cb219ff7eef480cfd074db2fac72","score":73.9,"scoreUnit":"percent","sourceUrl":"https://vulcanbench.com/benchmarks/14-grok-46-effort.html"},{"benchmarkId":"vulcanbench-v3","evidenceKind":"official_board","harnessId":"vulcanbench-v3-14-grok-46-effort-low","id":"s-vulcanbench-grok-4.6-14-grok-46-effort-low-da2f9d24acc8","ingestRunId":"ingest-vulcanbench-v3-da2f9d24acc8-fb878f2f","modelId":"grok-4.6","n":23,"observedOn":"2026-08-12","rawPayloadHash":"da2f9d24acc827209e42b26bace6a781f543cb219ff7eef480cfd074db2fac72","score":82.6,"scoreUnit":"percent","sourceUrl":"https://vulcanbench.com/benchmarks/14-grok-46-effort.html"},{"benchmarkId":"vulcanbench-v3","evidenceKind":"official_board","harnessId":"vulcanbench-v3-14-grok-46-effort-medium","id":"s-vulcanbench-grok-4.6-14-grok-46-effort-medium-da2f9d24acc8","ingestRunId":"ingest-vulcanbench-v3-da2f9d24acc8-fb878f2f","modelId":"grok-4.6","n":23,"observedOn":"2026-08-12","rawPayloadHash":"da2f9d24acc827209e42b26bace6a781f543cb219ff7eef480cfd074db2fac72","score":87,"scoreUnit":"percent","sourceUrl":"https://vulcanbench.com/benchmarks/14-grok-46-effort.html"},{"benchmarkId":"vulcanbench-v3","evidenceKind":"official_board","harnessId":"vulcanbench-v3-14-grok-46-effort-xhigh","id":"s-vulcanbench-grok-4.6-14-grok-46-effort-xhigh-da2f9d24acc8","ingestRunId":"ingest-vulcanbench-v3-da2f9d24acc8-fb878f2f","modelId":"grok-4.6","n":23,"observedOn":"2026-08-12","rawPayloadHash":"da2f9d24acc827209e42b26bace6a781f543cb219ff7eef480cfd074db2fac72","score":78.3,"scoreUnit":"percent","sourceUrl":"https://vulcanbench.com/benchmarks/14-grok-46-effort.html"},{"benchmarkId":"vulcanbench-v3","evidenceKind":"official_board","harnessId":"vulcanbench-v3-08-kimi-k3-max","id":"s-vulcanbench-kimi-k3-08-kimi-k3-max-da2f9d24acc8","ingestRunId":"ingest-vulcanbench-v3-da2f9d24acc8-fb878f2f","modelId":"kimi-k3","n":23,"observedOn":"2026-07-19","rawPayloadHash":"da2f9d24acc827209e42b26bace6a781f543cb219ff7eef480cfd074db2fac72","score":73.9,"scoreUnit":"percent","sourceUrl":"https://vulcanbench.com/benchmarks/08-kimi-k3.html"},{"benchmarkId":"vulcanbench-v3","evidenceKind":"official_board","harnessId":"vulcanbench-v3-19-musespark-effort-high","id":"s-vulcanbench-muse-spark-1.2-19-musespark-effort-high-da2f9d24acc8","ingestRunId":"ingest-vulcanbench-v3-da2f9d24acc8-fb878f2f","modelId":"muse-spark-1.2","n":23,"observedOn":"2026-08-25","rawPayloadHash":"da2f9d24acc827209e42b26bace6a781f543cb219ff7eef480cfd074db2fac72","score":73.9,"scoreUnit":"percent","sourceUrl":"https://vulcanbench.com/benchmarks/19-musespark-effort.html"},{"benchmarkId":"vulcanbench-v3","evidenceKind":"official_board","harnessId":"vulcanbench-v3-19-musespark-effort-low","id":"s-vulcanbench-muse-spark-1.2-19-musespark-effort-low-da2f9d24acc8","ingestRunId":"ingest-vulcanbench-v3-da2f9d24acc8-fb878f2f","modelId":"muse-spark-1.2","n":23,"observedOn":"2026-08-25","rawPayloadHash":"da2f9d24acc827209e42b26bace6a781f543cb219ff7eef480cfd074db2fac72","score":87,"scoreUnit":"percent","sourceUrl":"https://vulcanbench.com/benchmarks/19-musespark-effort.html"},{"benchmarkId":"vulcanbench-v3","evidenceKind":"official_board","harnessId":"vulcanbench-v3-19-musespark-effort-xhigh","id":"s-vulcanbench-muse-spark-1.2-19-musespark-effort-xhigh-da2f9d24acc8","ingestRunId":"ingest-vulcanbench-v3-da2f9d24acc8-fb878f2f","modelId":"muse-spark-1.2","n":23,"observedOn":"2026-08-25","rawPayloadHash":"da2f9d24acc827209e42b26bace6a781f543cb219ff7eef480cfd074db2fac72","score":52.2,"scoreUnit":"percent","sourceUrl":"https://vulcanbench.com/benchmarks/19-musespark-effort.html"},{"benchmarkId":"vulcanbench-v3","evidenceKind":"official_board","harnessId":"vulcanbench-v3-12-qwen38-max-low","id":"s-vulcanbench-qwen-3.8-max-12-qwen38-max-low-da2f9d24acc8","ingestRunId":"ingest-vulcanbench-v3-da2f9d24acc8-fb878f2f","modelId":"qwen-3.8-max","n":23,"observedOn":"2026-08-04","rawPayloadHash":"da2f9d24acc827209e42b26bace6a781f543cb219ff7eef480cfd074db2fac72","score":81.2,"scoreUnit":"percent","sourceUrl":"https://vulcanbench.com/benchmarks/12-qwen38-max.html"},{"benchmarkId":"vulcanbench-v3","evidenceKind":"official_board","harnessId":"vulcanbench-v3-12-qwen38-max-medium","id":"s-vulcanbench-qwen-3.8-max-12-qwen38-max-medium-da2f9d24acc8","ingestRunId":"ingest-vulcanbench-v3-da2f9d24acc8-fb878f2f","modelId":"qwen-3.8-max","n":23,"observedOn":"2026-08-04","rawPayloadHash":"da2f9d24acc827209e42b26bace6a781f543cb219ff7eef480cfd074db2fac72","score":71,"scoreUnit":"percent","sourceUrl":"https://vulcanbench.com/benchmarks/12-qwen38-max.html"},{"benchmarkId":"vulcanbench-v3","evidenceKind":"official_board","harnessId":"vulcanbench-v3-12-qwen38-max-xhigh","id":"s-vulcanbench-qwen-3.8-max-12-qwen38-max-xhigh-da2f9d24acc8","ingestRunId":"ingest-vulcanbench-v3-da2f9d24acc8-fb878f2f","modelId":"qwen-3.8-max","n":23,"observedOn":"2026-08-04","rawPayloadHash":"da2f9d24acc827209e42b26bace6a781f543cb219ff7eef480cfd074db2fac72","score":55.1,"scoreUnit":"percent","sourceUrl":"https://vulcanbench.com/benchmarks/12-qwen38-max.html"},{"benchmarkId":"vulcanbench-v3","evidenceKind":"official_board","harnessId":"vulcanbench-v3-17-qwen38-27b-effort-low","id":"s-vulcanbench-qwen3.8-27b-17-qwen38-27b-effort-low-da2f9d24acc8","ingestRunId":"ingest-vulcanbench-v3-da2f9d24acc8-fb878f2f","modelId":"qwen3.8-27b","n":23,"observedOn":"2026-08-21","rawPayloadHash":"da2f9d24acc827209e42b26bace6a781f543cb219ff7eef480cfd074db2fac72","score":82.6,"scoreUnit":"percent","sourceUrl":"https://vulcanbench.com/benchmarks/17-qwen38-27b-effort.html"},{"benchmarkId":"vulcanbench-v3","evidenceKind":"official_board","harnessId":"vulcanbench-v3-17-qwen38-27b-effort-medium","id":"s-vulcanbench-qwen3.8-27b-17-qwen38-27b-effort-medium-da2f9d24acc8","ingestRunId":"ingest-vulcanbench-v3-da2f9d24acc8-fb878f2f","modelId":"qwen3.8-27b","n":23,"observedOn":"2026-08-21","rawPayloadHash":"da2f9d24acc827209e42b26bace6a781f543cb219ff7eef480cfd074db2fac72","score":82.6,"scoreUnit":"percent","sourceUrl":"https://vulcanbench.com/benchmarks/17-qwen38-27b-effort.html"},{"benchmarkId":"vulcanbench-v3","evidenceKind":"official_board","harnessId":"vulcanbench-v3-17-qwen38-27b-effort-xhigh","id":"s-vulcanbench-qwen3.8-27b-17-qwen38-27b-effort-xhigh-da2f9d24acc8","ingestRunId":"ingest-vulcanbench-v3-da2f9d24acc8-fb878f2f","modelId":"qwen3.8-27b","n":23,"observedOn":"2026-08-21","rawPayloadHash":"da2f9d24acc827209e42b26bace6a781f543cb219ff7eef480cfd074db2fac72","score":73.9,"scoreUnit":"percent","sourceUrl":"https://vulcanbench.com/benchmarks/17-qwen38-27b-effort.html"}],"snapshot":{"createdAt":"2026-09-30T00:22:17.343Z","id":"sha256-40c5498eb0c4324d57e06cbe2cac1c74d3e63bd1c6b74d69daa043f780ac2cf2","note":"Model catalog reviewed 2026-09-29 across 24 labs. Existing sourced benchmark seed; additions have no inferred scores. AA Index is a labeled v4.1 snapshot, not a combiner input. · D1 scores"}}