{"version":"2026-09-09-effort-3-k26","snapshotId":"sha256-df858793dcb753b0f77beb223fce04e1de99a2c08f2c7b5f14534efcafa8c7c5","scope":"Observation-level labels from reviewed source configurations. Exact source, model and configuration matching, with benchmark locators when a footnote covers only part of a table. No API defaults inferred. Best across efforts is score selection, not the literal Max setting. Unmatched observations remain Not specified.","models":[{"modelId":"claude-fable-5","label":"Mixed settings","settings":[{"label":"Auto thinking","observations":2},{"label":"Max","observations":11},{"label":"Max with fallbacks","observations":45},{"label":"XHigh","observations":2},{"label":"XHigh with fallbacks","observations":1}],"unspecified":118,"total":179},{"modelId":"claude-fable-5-1","label":"Mixed settings","settings":[{"label":"Auto thinking","observations":2},{"label":"High","observations":1},{"label":"Max","observations":25},{"label":"Medium","observations":3},{"label":"XHigh","observations":4}],"unspecified":33,"total":68},{"modelId":"claude-haiku-4-5-20251001","label":"Not specified","settings":[],"unspecified":18,"total":18},{"modelId":"claude-opus-4-5-20251101","label":"Not specified","settings":[],"unspecified":3,"total":3},{"modelId":"claude-opus-4-6","label":"Max + unspecified","settings":[{"label":"Max","observations":32}],"unspecified":8,"total":40},{"modelId":"claude-opus-4-7","label":"Not specified","settings":[],"unspecified":33,"total":33},{"modelId":"claude-opus-4-8","label":"Mixed settings","settings":[{"label":"High","observations":1},{"label":"Max","observations":50}],"unspecified":106,"total":157},{"modelId":"claude-opus-5","label":"Mixed settings","settings":[{"label":"Auto thinking","observations":2},{"label":"Max","observations":19}],"unspecified":85,"total":106},{"modelId":"claude-sonnet-4-5-20250929","label":"Not specified","settings":[],"unspecified":33,"total":33},{"modelId":"claude-sonnet-4-6","label":"Maximum reasoning + unspecified","settings":[{"label":"Maximum reasoning","observations":11}],"unspecified":33,"total":44},{"modelId":"claude-sonnet-5","label":"Not specified","settings":[],"unspecified":15,"total":15},{"modelId":"deepseek-r1-0528","label":"Not specified","settings":[],"unspecified":3,"total":3},{"modelId":"deepseek-r1-0528-qwen3-8b","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"deepseek-r1-distill-llama-70b","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"deepseek-r1-distill-llama-8b","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"deepseek-r1-distill-qwen-1.5b","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"deepseek-r1-distill-qwen-14b","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"deepseek-r1-distill-qwen-32b","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"deepseek-r1-distill-qwen-7b","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"deepseek-v3.2","label":"Not specified","settings":[],"unspecified":3,"total":3},{"modelId":"deepseek-v3.2-speciale","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"deepseek-v4-flash","label":"Not specified","settings":[],"unspecified":13,"total":13},{"modelId":"deepseek-v4-flash-vision-exp","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"deepseek-v4-pro","label":"Max + unspecified","settings":[{"label":"Max","observations":4}],"unspecified":28,"total":32},{"modelId":"codegemma-7b-it","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"diffusiongemma-26b-a4b","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"gemini-2.5-flash","label":"Not specified","settings":[],"unspecified":22,"total":22},{"modelId":"gemini-2.5-flash-lite","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"gemini-2.5-pro","label":"Not specified","settings":[],"unspecified":24,"total":24},{"modelId":"gemini-3-flash-preview","label":"Not specified","settings":[],"unspecified":2,"total":2},{"modelId":"gemini-3.1-flash-lite","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"gemini-3.1-pro","label":"High + unspecified","settings":[{"label":"High","observations":31}],"unspecified":78,"total":109},{"modelId":"gemini-3.5-flash","label":"Not specified","settings":[],"unspecified":6,"total":6},{"modelId":"gemini-3.5-flash-lite","label":"Not specified","settings":[],"unspecified":4,"total":4},{"modelId":"gemini-3.6-flash","label":"Not specified","settings":[],"unspecified":2,"total":2},{"modelId":"gemini-3.7-flash","label":"Not specified","settings":[],"unspecified":17,"total":17},{"modelId":"gemini-3.8-flash","label":"High thinking + unspecified","settings":[{"label":"High thinking","observations":1}],"unspecified":25,"total":26},{"modelId":"gemma-1-2b-it","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"gemma-1-7b-it","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"gemma-1.1-2b-it","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"gemma-1.1-7b-it","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"gemma-2-27b-it","label":"Not specified","settings":[],"unspecified":2,"total":2},{"modelId":"gemma-2-2b-it","label":"Not specified","settings":[],"unspecified":2,"total":2},{"modelId":"gemma-2-9b-it","label":"Not specified","settings":[],"unspecified":2,"total":2},{"modelId":"gemma-3-12b-it","label":"Not specified","settings":[],"unspecified":2,"total":2},{"modelId":"gemma-3-1b-it","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"gemma-3-270m-it","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"gemma-3-27b-it","label":"Not specified","settings":[],"unspecified":2,"total":2},{"modelId":"gemma-3-4b-it","label":"Not specified","settings":[],"unspecified":2,"total":2},{"modelId":"gemma-3n-e2b-it","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"gemma-3n-e4b-it","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"gemma-4-12b-it","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"gemma-4-26b-a4b-it","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"gemma-4-31b-it","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"gemma-4-e2b-it","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"gemma-4-e4b-it","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"recurrentgemma-2b-it","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"recurrentgemma-9b-it","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"llama-3.1-405b-instruct","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"llama-3.1-70b-instruct","label":"Not specified","settings":[],"unspecified":2,"total":2},{"modelId":"llama-3.1-8b-instruct","label":"Not specified","settings":[],"unspecified":2,"total":2},{"modelId":"llama-3.2-11b-vision-instruct","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"llama-3.2-1b-instruct","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"llama-3.2-3b-instruct","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"llama-3.2-90b-vision-instruct","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"llama-3.3-70b-instruct","label":"Not specified","settings":[],"unspecified":2,"total":2},{"modelId":"llama-4-maverick","label":"Not specified","settings":[],"unspecified":8,"total":8},{"modelId":"llama-4-scout","label":"Not specified","settings":[],"unspecified":2,"total":2},{"modelId":"muse-glimmer-30b","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"muse-spark-1.1","label":"Not specified","settings":[],"unspecified":23,"total":23},{"modelId":"muse-spark-1.2","label":"XHigh + unspecified","settings":[{"label":"XHigh","observations":12}],"unspecified":4,"total":16},{"modelId":"muse-spark-1.3","label":"Mixed settings","settings":[{"label":"Max","observations":12},{"label":"XHigh","observations":11}],"unspecified":1,"total":24},{"modelId":"codestral-2508","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"devstral-2512","label":"Not specified","settings":[],"unspecified":3,"total":3},{"modelId":"devstral-medium-2507","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"devstral-small-2507","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"labs-devstral-small-2512","label":"Not specified","settings":[],"unspecified":2,"total":2},{"modelId":"labs-mistral-small-creative","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"leanstral-1.5","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"magistral-medium-2509","label":"Maximum reasoning + unspecified","settings":[{"label":"Maximum reasoning","observations":5}],"unspecified":1,"total":6},{"modelId":"magistral-small-2509","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"ministral-3-14b","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"ministral-3-3b","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"ministral-3-8b","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"ministral-3b-2410","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"ministral-8b-2410","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"mistral-large-2411","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"mistral-large-3","label":"Not specified","settings":[],"unspecified":3,"total":3},{"modelId":"mistral-medium-2505","label":"Not specified","settings":[],"unspecified":2,"total":2},{"modelId":"mistral-medium-2508","label":"Maximum reasoning + unspecified","settings":[{"label":"Maximum reasoning","observations":5}],"unspecified":2,"total":7},{"modelId":"mistral-medium-3.5","label":"Maximum reasoning + unspecified","settings":[{"label":"Maximum reasoning","observations":10}],"unspecified":3,"total":13},{"modelId":"mistral-small-2506","label":"Not specified","settings":[],"unspecified":2,"total":2},{"modelId":"mistral-small-4","label":"Maximum reasoning","settings":[{"label":"Maximum reasoning","observations":5}],"unspecified":0,"total":5},{"modelId":"open-mistral-nemo-2407","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"pixtral-12b-2409","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"pixtral-large-2411","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"voxtral-mini-2507","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"voxtral-small","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"kimi-dev-72b","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"kimi-k2.6","label":"Reasoning + unspecified","settings":[{"label":"Reasoning","observations":33}],"unspecified":53,"total":86},{"modelId":"kimi-k2.7-code","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"kimi-k3","label":"Max + unspecified","settings":[{"label":"Max","observations":52}],"unspecified":34,"total":86},{"modelId":"kimi-linear-48b-a3b-instruct","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"kimi-vl-a3b-instruct","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"kimi-vl-a3b-thinking-2506","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"nemotron-3-nano-omni-30b-a3b-reasoning-bf16","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"nemotron-3-ultra","label":"Thinking + unspecified","settings":[{"label":"Thinking","observations":12}],"unspecified":33,"total":45},{"modelId":"nvidia-nemotron-3-nano-30b-a3b-bf16","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"nvidia-nemotron-3-nano-4b-bf16","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"nvidia-nemotron-3-super-120b-a12b-bf16","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"nvidia-nemotron-3.5-lightning-30b-a3b-bf16","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"nvidia-nemotron-nano-12b-v2","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"nvidia-nemotron-nano-12b-v2-vl-bf16","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"nvidia-nemotron-nano-9b-v2","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"nvidia-nemotron-nano-9b-v2-japanese","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"gpt-3.5-turbo","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"gpt-3.5-turbo-1106","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"gpt-3.5-turbo-instruct","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"gpt-4","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"gpt-4-turbo","label":"Not specified","settings":[],"unspecified":2,"total":2},{"modelId":"gpt-4.1","label":"Not specified","settings":[],"unspecified":3,"total":3},{"modelId":"gpt-4.1-mini","label":"Not specified","settings":[],"unspecified":2,"total":2},{"modelId":"gpt-4.1-nano","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"gpt-4o","label":"Not specified","settings":[],"unspecified":2,"total":2},{"modelId":"gpt-4o-mini","label":"Not specified","settings":[],"unspecified":4,"total":4},{"modelId":"gpt-5","label":"Not specified","settings":[],"unspecified":18,"total":18},{"modelId":"gpt-5-mini","label":"Not specified","settings":[],"unspecified":18,"total":18},{"modelId":"gpt-5-nano","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"gpt-5-pro","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"gpt-5.1","label":"Not specified","settings":[],"unspecified":18,"total":18},{"modelId":"gpt-5.2","label":"Not specified","settings":[],"unspecified":3,"total":3},{"modelId":"gpt-5.2-pro","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"gpt-5.3-codex","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"gpt-5.3-codex-spark","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"gpt-5.4","label":"XHigh + unspecified","settings":[{"label":"XHigh","observations":28}],"unspecified":8,"total":36},{"modelId":"gpt-5.4-mini","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"gpt-5.4-nano","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"gpt-5.4-pro","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"gpt-5.5","label":"XHigh + unspecified","settings":[{"label":"XHigh","observations":49}],"unspecified":99,"total":148},{"modelId":"gpt-5.5-pro","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"gpt-5.6-luna","label":"Not specified","settings":[],"unspecified":50,"total":50},{"modelId":"gpt-5.6-sol","label":"Mixed settings","settings":[{"label":"Best across efforts","observations":33},{"label":"Max","observations":81}],"unspecified":133,"total":247},{"modelId":"gpt-5.6-terra","label":"Not specified","settings":[],"unspecified":64,"total":64},{"modelId":"gpt-6-astra","label":"Best across efforts + unspecified","settings":[{"label":"Best across efforts","observations":34}],"unspecified":26,"total":60},{"modelId":"gpt-oss-120b","label":"Not specified","settings":[],"unspecified":3,"total":3},{"modelId":"gpt-oss-20b","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"o1","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"o1-pro","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"o3","label":"Not specified","settings":[],"unspecified":4,"total":4},{"modelId":"o3-mini","label":"Not specified","settings":[],"unspecified":3,"total":3},{"modelId":"o3-pro","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"o4-mini","label":"Not specified","settings":[],"unspecified":3,"total":3},{"modelId":"qvq-max","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qvq-plus","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qwen-3.8-max","label":"Max + unspecified","settings":[{"label":"Max","observations":2}],"unspecified":56,"total":58},{"modelId":"qwen-coder-plus","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qwen-coder-turbo","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qwen-flash","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qwen-flash-character","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qwen-long","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qwen-math-plus","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qwen-math-turbo","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qwen-max","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qwen-omni-turbo","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qwen-plus","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qwen-plus-character","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qwen-plus-character-ja","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qwen-turbo","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qwen-vl-max","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qwen-vl-plus","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qwen3-0.6b","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qwen3-1.7b","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qwen3-14b","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qwen3-235b-a22b","label":"Not specified","settings":[],"unspecified":3,"total":3},{"modelId":"qwen3-235b-a22b-instruct-2507","label":"Not specified","settings":[],"unspecified":3,"total":3},{"modelId":"qwen3-235b-a22b-thinking-2507","label":"Not specified","settings":[],"unspecified":2,"total":2},{"modelId":"qwen3-30b-a3b","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"qwen3-30b-a3b-instruct-2507","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"qwen3-30b-a3b-thinking-2507","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"qwen3-32b","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"qwen3-4b","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qwen3-4b-instruct-2507","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qwen3-4b-thinking-2507","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qwen3-8b","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qwen3-coder-30b-a3b-instruct","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qwen3-coder-480b-a35b-instruct","label":"Not specified","settings":[],"unspecified":2,"total":2},{"modelId":"qwen3-coder-flash","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qwen3-coder-next","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qwen3-coder-plus","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qwen3-max","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qwen3-next-80b-a3b-instruct","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"qwen3-next-80b-a3b-thinking","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"qwen3-omni-30b-a3b-instruct","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qwen3-omni-30b-a3b-thinking","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qwen3-omni-flash","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qwen3-vl-235b-a22b-instruct","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"qwen3-vl-235b-a22b-thinking","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"qwen3-vl-2b-instruct","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qwen3-vl-2b-thinking","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qwen3-vl-30b-a3b-instruct","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qwen3-vl-30b-a3b-thinking","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qwen3-vl-32b-instruct","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qwen3-vl-32b-thinking","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qwen3-vl-4b-instruct","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qwen3-vl-4b-thinking","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qwen3-vl-8b-instruct","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qwen3-vl-8b-thinking","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qwen3-vl-flash","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qwen3-vl-plus","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qwen3.5-0.8b","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"qwen3.5-122b-a10b","label":"Not specified","settings":[],"unspecified":2,"total":2},{"modelId":"qwen3.5-27b","label":"Not specified","settings":[],"unspecified":2,"total":2},{"modelId":"qwen3.5-2b","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"qwen3.5-35b-a3b","label":"Not specified","settings":[],"unspecified":2,"total":2},{"modelId":"qwen3.5-397b-a17b","label":"Not specified","settings":[],"unspecified":46,"total":46},{"modelId":"qwen3.5-4b","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"qwen3.5-9b","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"qwen3.5-flash","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"qwen3.5-omni-flash","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qwen3.5-omni-plus","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qwen3.5-plus","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qwen3.6-27b","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qwen3.6-35b-a3b","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qwen3.6-flash","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qwen3.6-plus","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"qwen3.7-flash","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qwen3.7-max","label":"Not specified","settings":[],"unspecified":32,"total":32},{"modelId":"qwen3.7-plus","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"qwen3.8-2.4t-a95b","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"qwen3.8-27b","label":"Not specified","settings":[],"unspecified":5,"total":5},{"modelId":"qwen3.8-flash","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qwen3.8-flash-next","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"qwq-plus","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"grok-4.20-0309-non-reasoning","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"grok-4.20-0309-reasoning","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"grok-4.20-multi-agent-0309","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"grok-4.3","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"grok-4.5","label":"Not specified","settings":[],"unspecified":15,"total":15},{"modelId":"grok-4.6","label":"Not specified","settings":[],"unspecified":21,"total":21},{"modelId":"grok-build-0.1","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"glm-5.3-flash","label":"Not specified","settings":[],"unspecified":4,"total":4},{"modelId":"glm-5.3","label":"Max + unspecified","settings":[{"label":"Max","observations":9}],"unspecified":13,"total":22},{"modelId":"glm-5.2","label":"Max + unspecified","settings":[{"label":"Max","observations":28}],"unspecified":29,"total":57},{"modelId":"glm-5.1","label":"Not specified","settings":[],"unspecified":54,"total":54},{"modelId":"glm-5","label":"Not specified","settings":[],"unspecified":7,"total":7},{"modelId":"glm-4.7","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"glm-4.7-flashx","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"glm-4.6","label":"Not specified","settings":[],"unspecified":2,"total":2},{"modelId":"glm-4.5","label":"Not specified","settings":[],"unspecified":3,"total":3},{"modelId":"glm-4.5-x","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"glm-4.5-air","label":"Not specified","settings":[],"unspecified":2,"total":2},{"modelId":"glm-4.5-airx","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"glm-4-32b-0414-128k","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"glm-4.7-flash","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"glm-4.5-flash","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"glm-4.6v","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"glm-4.6v-flashx","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"glm-4.5v","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"glm-4.6v-flash","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"glm-5-turbo","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"glm-5v-turbo","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"glm-4.1v-9b-thinking","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"glm-z1-rumination-32b-0414","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"glm-z1-9b-0414","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"glm-z1-32b-0414","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"glm-4-9b-0414","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"glm-4-32b-0414","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"glm-4-9b-chat","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"glm-4-9b-chat-1m","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"glm-4v-9b","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"glm-edge-1.5b-chat","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"glm-edge-4b-chat","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"glm-edge-v-2b","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"glm-edge-v-5b","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"chatglm-6b","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"chatglm2-6b","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"chatglm2-6b-32k","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"chatglm3-6b","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"chatglm3-6b-32k","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"chatglm3-6b-128k","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"codegeex4-all-9b","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"autoglm-phone-9b","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"autoglm-phone-9b-multilingual","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"cogagent-9b-20241220","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"codegeex2-6b","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"visualglm-6b","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"cogvlm-chat-hf","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"cogagent-chat-hf","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"cogagent-vqa-hf","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"cogvlm2-llama3-chat-19b","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"cogvlm2-llama3-chinese-chat-19b","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"cogvlm2-video-llama3-chat","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"minimax-m3","label":"Not specified","settings":[],"unspecified":35,"total":35},{"modelId":"minimax-m2.7","label":"Not specified","settings":[],"unspecified":55,"total":55},{"modelId":"minimax-m2.5","label":"Not specified","settings":[],"unspecified":4,"total":4},{"modelId":"minimax-m2.1","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"minimax-m2","label":"Not specified","settings":[],"unspecified":3,"total":3},{"modelId":"minimax-text-01","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"minimax-vl-01","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"minimax-m1-40k","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"minimax-m1-80k","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"synlogic-32b","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"synlogic-mix-3-32b","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"synlogic-7b","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"minimax-m2-her","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"command-a-plus-05-2026","label":"Not specified","settings":[],"unspecified":16,"total":16},{"modelId":"command-a-03-2025","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"command-r7b-12-2024","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"command-a-translate-08-2025","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"command-a-reasoning-08-2025","label":"Not specified","settings":[],"unspecified":10,"total":10},{"modelId":"command-a-vision-07-2025","label":"Not specified","settings":[],"unspecified":4,"total":4},{"modelId":"command-r-08-2024","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"command-r-plus-08-2024","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"command-r-03-2024","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"command-r-plus-04-2024","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"command-light","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"command","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"tiny-aya-global","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"tiny-aya-earth","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"tiny-aya-fire","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"tiny-aya-water","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"c4ai-aya-expanse-32b","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"c4ai-aya-vision-32b","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"c4ai-command-r7b-arabic-02-2025","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"c4ai-aya-expanse-8b","label":"Not specified","settings":[],"unspecified":2,"total":2},{"modelId":"c4ai-aya-vision-8b","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"aya-23-8b","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"aya-23-35b","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"aya-101","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"north-mini-code-1-0","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"north-micro-vision-instruct","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"phi-4","label":"Not specified","settings":[],"unspecified":2,"total":2},{"modelId":"phi-4-mini-instruct","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"phi-4-reasoning-vision-15b","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"phi-3-mini-4k-instruct","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"phi-3-mini-128k-instruct","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"phi-3.5-mini-instruct","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"phi-4-multimodal-instruct","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"phi-4-reasoning","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"phi-mini-moe-instruct","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"phi-tiny-moe-instruct","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"phi-3-medium-4k-instruct","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"phi-3-medium-128k-instruct","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"phi-3-small-8k-instruct","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"phi-3-small-128k-instruct","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"phi-3-vision-128k-instruct","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"phi-3.5-vision-instruct","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"phi-3.5-moe-instruct","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"phi-4-reasoning-plus","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"phi-4-mini-reasoning","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"phi-4-mini-flash-reasoning","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"mai-ds-r1","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"mai-thinking-1","label":"Not specified","settings":[],"unspecified":19,"total":19},{"modelId":"mai-code-1-1-flash","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"mai-code-1-flash","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"amazon-nova-micro","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"amazon-nova-lite","label":"Not specified","settings":[],"unspecified":12,"total":12},{"modelId":"amazon-nova-pro","label":"Not specified","settings":[],"unspecified":12,"total":12},{"modelId":"amazon-nova-premier","label":"Not specified","settings":[],"unspecified":16,"total":16},{"modelId":"amazon-nova-2-lite","label":"Not specified","settings":[],"unspecified":20,"total":20},{"modelId":"amazon-nova-2-pro-preview","label":"Not specified","settings":[],"unspecified":19,"total":19},{"modelId":"deepseek-v4-pro-0424","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"doubao-seed-1.6","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"doubao-seed-1.6-thinking","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"doubao-seed-1.6-vision","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"doubao-seed-1.8","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"doubao-seed-2.0-code","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"doubao-seed-2.0-lite","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"doubao-seed-2.0-mini","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"doubao-seed-2.0-pro","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"doubao-seed-2.1-pro","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"doubao-seed-2.1-turbo","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"doubao-seed-character","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"doubao-seed-code","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"doubao-seed-evolving","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"seed-coder-8b-instruct","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"seed-coder-8b-reasoning","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"seed-oss-36b-instruct","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"exaone-3.0-7.8b-instruct","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"exaone-3.5-2.4b-instruct","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"exaone-3.5-32b-instruct","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"exaone-3.5-7.8b-instruct","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"exaone-4.0-1.2b","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"exaone-4.0-32b","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"exaone-4.0.1-32b","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"exaone-4.5-33b","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"exaone-deep-2.4b","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"exaone-deep-32b","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"exaone-deep-7.8b","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"k-exaone-2.0-750b-a37b","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"k-exaone-236b-a23b","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"step-3.5-flash","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"step-3.7-flash","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"step3","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"step3-vl-10b","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"solar-10.7b-instruct-v1.0","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"solar-mini-250422","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"solar-open-100b","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"solar-open2-250b","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"solar-pro-preview-instruct","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"solar-pro2-251215","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"solar-pro3-260323","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"solar-pro4-260806","label":"Not specified","settings":[],"unspecified":2,"total":2},{"modelId":"mimo-7b-rl","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"mimo-7b-rl-0530","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"mimo-7b-rl-zero","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"mimo-7b-sft","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"mimo-v2-flash","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"mimo-v2.5","label":"Not specified","settings":[],"unspecified":2,"total":2},{"modelId":"mimo-v2.5-pro","label":"Not specified","settings":[],"unspecified":2,"total":2},{"modelId":"mimo-vl-7b-rl","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"mimo-vl-7b-rl-2508","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"mimo-vl-7b-sft","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"mimo-vl-7b-sft-2508","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"ernie-4.5-0.3b-pt","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"ernie-4.5-21b-a3b-pt","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"ernie-4.5-21b-a3b-thinking","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"ernie-4.5-300b-a47b-pt","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"ernie-4.5-vl-28b-a3b-pt","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"ernie-4.5-vl-28b-a3b-thinking","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"ernie-4.5-vl-424b-a47b-pt","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qianfan-vl-3b","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qianfan-vl-70b","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"qianfan-vl-8b","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"hunyuan-0.5b-instruct","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"hunyuan-1.8b-instruct","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"hunyuan-4b-instruct","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"hunyuan-7b-instruct","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"hunyuan-7b-instruct-0124","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"hunyuan-a13b-instruct","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"hy3","label":"Not specified","settings":[],"unspecified":2,"total":2},{"modelId":"hy3-preview","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"hy4-preview","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"hunyuan-a52b-instruct","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"wedlm-7b-instruct","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"wedlm-8b-instruct","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"youtu-llm-2b","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"ernie-5.1","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"ernie-5.0","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"ernie-5.0-thinking-preview","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"ernie-5.0-thinking-latest","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"ernie-5.0-thinking-exp","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"ernie-4.5-turbo-32k","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"ernie-4.5-turbo-128k","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"ernie-4.5-turbo-20260402","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"ernie-4.5-turbo-vl","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"ernie-4.5-turbo-vl-32k","label":"No results","settings":[],"unspecified":0,"total":0},{"modelId":"minicpm5-2b","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"agnes-2.5-pro-alpha","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"mistral-large-2407","label":"Not specified","settings":[],"unspecified":1,"total":1},{"modelId":"deepseek-v4-flash-0424","label":"No results","settings":[],"unspecified":0,"total":0}],"observations":[{"observationId":"launch-amazon-1491","modelId":"amazon-nova-lite","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1492","modelId":"amazon-nova-lite","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1493","modelId":"amazon-nova-lite","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1494","modelId":"amazon-nova-lite","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1495","modelId":"amazon-nova-lite","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1496","modelId":"claude-haiku-4-5-20251001","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1497","modelId":"claude-haiku-4-5-20251001","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1498","modelId":"claude-haiku-4-5-20251001","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1499","modelId":"claude-haiku-4-5-20251001","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1500","modelId":"claude-haiku-4-5-20251001","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1501","modelId":"gpt-5-mini","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1502","modelId":"gpt-5-mini","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1503","modelId":"gpt-5-mini","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1504","modelId":"gpt-5-mini","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1505","modelId":"gpt-5-mini","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1506","modelId":"gemini-2.5-flash","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1507","modelId":"gemini-2.5-flash","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1508","modelId":"gemini-2.5-flash","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1509","modelId":"gemini-2.5-flash","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1510","modelId":"gemini-2.5-flash","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1511","modelId":"gemini-2.5-flash","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1512","modelId":"amazon-nova-2-lite","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1513","modelId":"amazon-nova-2-lite","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1514","modelId":"amazon-nova-2-lite","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1515","modelId":"amazon-nova-2-lite","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1516","modelId":"amazon-nova-2-lite","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1517","modelId":"amazon-nova-2-lite","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1518","modelId":"amazon-nova-pro","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1519","modelId":"amazon-nova-pro","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1520","modelId":"amazon-nova-pro","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1521","modelId":"amazon-nova-pro","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1522","modelId":"amazon-nova-pro","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1523","modelId":"amazon-nova-premier","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1524","modelId":"amazon-nova-premier","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1525","modelId":"amazon-nova-premier","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1526","modelId":"amazon-nova-premier","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1527","modelId":"amazon-nova-premier","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1528","modelId":"amazon-nova-premier","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1529","modelId":"claude-sonnet-4-5-20250929","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1530","modelId":"claude-sonnet-4-5-20250929","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1531","modelId":"claude-sonnet-4-5-20250929","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1532","modelId":"claude-sonnet-4-5-20250929","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1533","modelId":"claude-sonnet-4-5-20250929","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1534","modelId":"claude-sonnet-4-5-20250929","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1535","modelId":"gpt-5","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1536","modelId":"gpt-5","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1537","modelId":"gpt-5","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1538","modelId":"gpt-5","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1539","modelId":"gpt-5","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1540","modelId":"gpt-5.1","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1541","modelId":"gpt-5.1","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1542","modelId":"gpt-5.1","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1543","modelId":"gpt-5.1","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1544","modelId":"gpt-5.1","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1545","modelId":"gemini-2.5-pro","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1546","modelId":"gemini-2.5-pro","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1547","modelId":"gemini-2.5-pro","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1548","modelId":"gemini-2.5-pro","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1549","modelId":"gemini-2.5-pro","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1550","modelId":"gemini-2.5-pro","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1551","modelId":"amazon-nova-2-pro-preview","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1552","modelId":"amazon-nova-2-pro-preview","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1553","modelId":"amazon-nova-2-pro-preview","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1554","modelId":"amazon-nova-2-pro-preview","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1555","modelId":"amazon-nova-2-pro-preview","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1556","modelId":"amazon-nova-2-pro-preview","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1557","modelId":"amazon-nova-lite","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1558","modelId":"amazon-nova-lite","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1559","modelId":"claude-haiku-4-5-20251001","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1560","modelId":"claude-haiku-4-5-20251001","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1561","modelId":"claude-haiku-4-5-20251001","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1562","modelId":"claude-haiku-4-5-20251001","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1563","modelId":"gpt-5-mini","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1564","modelId":"gpt-5-mini","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1565","modelId":"gpt-5-mini","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1566","modelId":"gpt-5-mini","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1567","modelId":"gpt-5-mini","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1568","modelId":"gemini-2.5-flash","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1569","modelId":"gemini-2.5-flash","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1570","modelId":"gemini-2.5-flash","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1571","modelId":"gemini-2.5-flash","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1572","modelId":"amazon-nova-2-lite","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1573","modelId":"amazon-nova-2-lite","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1574","modelId":"amazon-nova-2-lite","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1575","modelId":"amazon-nova-2-lite","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1576","modelId":"amazon-nova-2-lite","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1577","modelId":"amazon-nova-pro","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1578","modelId":"amazon-nova-pro","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1579","modelId":"amazon-nova-premier","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1580","modelId":"amazon-nova-premier","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1581","modelId":"claude-sonnet-4-5-20250929","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1582","modelId":"claude-sonnet-4-5-20250929","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1583","modelId":"claude-sonnet-4-5-20250929","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1584","modelId":"claude-sonnet-4-5-20250929","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1585","modelId":"claude-sonnet-4-5-20250929","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1586","modelId":"gpt-5","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1587","modelId":"gpt-5","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1588","modelId":"gpt-5","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1589","modelId":"gpt-5","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1590","modelId":"gpt-5","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1591","modelId":"gpt-5.1","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1592","modelId":"gpt-5.1","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1593","modelId":"gpt-5.1","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1594","modelId":"gpt-5.1","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1595","modelId":"gemini-2.5-pro","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1596","modelId":"gemini-2.5-pro","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1597","modelId":"gemini-2.5-pro","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1598","modelId":"gemini-2.5-pro","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1599","modelId":"gemini-2.5-pro","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1600","modelId":"amazon-nova-2-pro-preview","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1601","modelId":"amazon-nova-2-pro-preview","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1602","modelId":"amazon-nova-2-pro-preview","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1603","modelId":"amazon-nova-2-pro-preview","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1604","modelId":"amazon-nova-lite","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1605","modelId":"amazon-nova-lite","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1606","modelId":"amazon-nova-lite","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1607","modelId":"amazon-nova-lite","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1608","modelId":"amazon-nova-lite","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1609","modelId":"claude-haiku-4-5-20251001","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1610","modelId":"claude-haiku-4-5-20251001","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1611","modelId":"claude-haiku-4-5-20251001","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1612","modelId":"gpt-5-mini","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1613","modelId":"gpt-5-mini","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1614","modelId":"gpt-5-mini","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1615","modelId":"gpt-5-mini","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1616","modelId":"gemini-2.5-flash","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1617","modelId":"gemini-2.5-flash","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1618","modelId":"gemini-2.5-flash","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1619","modelId":"gemini-2.5-flash","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1620","modelId":"gemini-2.5-flash","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1621","modelId":"amazon-nova-2-lite","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1622","modelId":"amazon-nova-2-lite","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1623","modelId":"amazon-nova-2-lite","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1624","modelId":"amazon-nova-2-lite","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1625","modelId":"amazon-nova-2-lite","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1626","modelId":"amazon-nova-pro","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1627","modelId":"amazon-nova-pro","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1628","modelId":"amazon-nova-pro","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1629","modelId":"amazon-nova-pro","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1630","modelId":"amazon-nova-pro","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1631","modelId":"amazon-nova-premier","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1632","modelId":"amazon-nova-premier","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1633","modelId":"amazon-nova-premier","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1634","modelId":"amazon-nova-premier","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1635","modelId":"amazon-nova-premier","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1636","modelId":"claude-sonnet-4-5-20250929","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1637","modelId":"claude-sonnet-4-5-20250929","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1638","modelId":"claude-sonnet-4-5-20250929","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1639","modelId":"gpt-5","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1640","modelId":"gpt-5","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1641","modelId":"gpt-5","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1642","modelId":"gpt-5","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1643","modelId":"gpt-5.1","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1644","modelId":"gpt-5.1","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1645","modelId":"gpt-5.1","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1646","modelId":"gpt-5.1","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1647","modelId":"gemini-2.5-pro","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1648","modelId":"gemini-2.5-pro","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1649","modelId":"gemini-2.5-pro","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1650","modelId":"gemini-2.5-pro","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1651","modelId":"gemini-2.5-pro","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1652","modelId":"amazon-nova-2-pro-preview","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1653","modelId":"amazon-nova-2-pro-preview","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1654","modelId":"amazon-nova-2-pro-preview","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1655","modelId":"amazon-nova-2-pro-preview","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1656","modelId":"amazon-nova-2-pro-preview","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 launch evaluation; benchmark-specific settings in report. tau2 Verified avg@3; telecom cited AA because user model sensitivity; comparators mix own runs and cited provider scores."},{"observationId":"launch-amazon-1657","modelId":"claude-haiku-4-5-20251001","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report."},{"observationId":"launch-amazon-1658","modelId":"claude-haiku-4-5-20251001","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report."},{"observationId":"launch-amazon-1659","modelId":"claude-haiku-4-5-20251001","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report."},{"observationId":"launch-amazon-1660","modelId":"gpt-5-mini","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report."},{"observationId":"launch-amazon-1661","modelId":"gpt-5-mini","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report."},{"observationId":"launch-amazon-1662","modelId":"gpt-5-mini","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report."},{"observationId":"launch-amazon-1663","modelId":"gemini-2.5-flash","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report."},{"observationId":"launch-amazon-1664","modelId":"gemini-2.5-flash","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report."},{"observationId":"launch-amazon-1665","modelId":"gemini-2.5-flash","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report."},{"observationId":"launch-amazon-1666","modelId":"gemini-2.5-flash","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report."},{"observationId":"launch-amazon-1667","modelId":"amazon-nova-2-lite","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report."},{"observationId":"launch-amazon-1668","modelId":"amazon-nova-2-lite","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report."},{"observationId":"launch-amazon-1669","modelId":"amazon-nova-2-lite","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report."},{"observationId":"launch-amazon-1670","modelId":"amazon-nova-2-lite","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report."},{"observationId":"launch-amazon-1671","modelId":"amazon-nova-premier","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report."},{"observationId":"launch-amazon-1672","modelId":"amazon-nova-premier","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report."},{"observationId":"launch-amazon-1673","modelId":"amazon-nova-premier","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report."},{"observationId":"launch-amazon-1674","modelId":"claude-sonnet-4-5-20250929","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report."},{"observationId":"launch-amazon-1675","modelId":"claude-sonnet-4-5-20250929","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report."},{"observationId":"launch-amazon-1676","modelId":"claude-sonnet-4-5-20250929","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report."},{"observationId":"launch-amazon-1677","modelId":"claude-sonnet-4-5-20250929","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report."},{"observationId":"launch-amazon-1678","modelId":"gpt-5","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report."},{"observationId":"launch-amazon-1679","modelId":"gpt-5","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report."},{"observationId":"launch-amazon-1680","modelId":"gpt-5","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report."},{"observationId":"launch-amazon-1681","modelId":"gpt-5.1","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report."},{"observationId":"launch-amazon-1682","modelId":"gpt-5.1","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report."},{"observationId":"launch-amazon-1683","modelId":"gpt-5.1","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report."},{"observationId":"launch-amazon-1684","modelId":"gemini-2.5-pro","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report."},{"observationId":"launch-amazon-1685","modelId":"gemini-2.5-pro","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report."},{"observationId":"launch-amazon-1686","modelId":"gemini-2.5-pro","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report."},{"observationId":"launch-amazon-1687","modelId":"gemini-2.5-pro","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report."},{"observationId":"launch-amazon-1688","modelId":"amazon-nova-2-pro-preview","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report."},{"observationId":"launch-amazon-1689","modelId":"amazon-nova-2-pro-preview","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report."},{"observationId":"launch-amazon-1690","modelId":"amazon-nova-2-pro-preview","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report."},{"observationId":"launch-amazon-1691","modelId":"amazon-nova-2-pro-preview","label":"Not specified","sourceUrl":"https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf","configuration":"Nova2 internal agentic scaffold; scaled inference explicitly separate. Comparator setups from cited provider report."},{"observationId":"launch-anthropic-fable-5-1-card-0","modelId":"claude-fable-5-1","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Anthropic reported configuration; adaptive thinking at max effort, default sampling, five-trial mean unless section specifies otherwise; context at most 1M. Comparator configuration follows cited prior card or board. Agent identity is not established as mini-swe-agent."},{"observationId":"launch-anthropic-fable-5-1-card-1","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Anthropic reported configuration; adaptive thinking at max effort, default sampling, five-trial mean unless section specifies otherwise; context at most 1M. Comparator configuration follows cited prior card or board. Agent identity is not established as mini-swe-agent."},{"observationId":"launch-anthropic-fable-5-1-card-10","modelId":"claude-fable-5-1","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Anthropic reported configuration; adaptive thinking at max effort, default sampling, five-trial mean unless section specifies otherwise; context at most 1M. Comparator configuration follows cited prior card or board. 113 tasks; original hidden-test grading."},{"observationId":"launch-anthropic-fable-5-1-card-100","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"ARC Prize semi-private validation; verified Fable5.1 max effort; comparator figures as reported in summary."},{"observationId":"launch-anthropic-fable-5-1-card-101","modelId":"claude-fable-5-1","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"ARC Prize semi-private validation; verified Fable5.1 max effort; comparator figures as reported in summary."},{"observationId":"launch-anthropic-fable-5-1-card-102","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"ARC Prize semi-private validation; verified Fable5.1 max effort; comparator figures as reported in summary."},{"observationId":"launch-anthropic-fable-5-1-card-103","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"ARC Prize semi-private validation; verified Fable5.1 max effort; comparator figures as reported in summary."},{"observationId":"launch-anthropic-fable-5-1-card-104","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"ARC Prize semi-private validation; verified Fable5.1 max effort; comparator figures as reported in summary."},{"observationId":"launch-anthropic-fable-5-1-card-105","modelId":"claude-fable-5-1","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Raw rubric score; adaptive max; five trials; no tools/custom system prompt; Opus4.8grader; Fable5.1 safety fallback toOpus5."},{"observationId":"launch-anthropic-fable-5-1-card-106","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Raw rubric score; adaptive max; five trials; no tools/custom system prompt; Opus4.8grader; Fable5.1 safety fallback toOpus5."},{"observationId":"launch-anthropic-fable-5-1-card-107","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Raw rubric score; adaptive max; five trials; no tools/custom system prompt; Opus4.8grader; Fable5.1 safety fallback toOpus5."},{"observationId":"launch-anthropic-fable-5-1-card-108","modelId":"claude-fable-5-1","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Raw rubric score; adaptive max; five trials; no tools/custom system prompt; Opus4.8grader; Fable5.1 safety fallback toOpus5."},{"observationId":"launch-anthropic-fable-5-1-card-109","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Raw rubric score; adaptive max; five trials; no tools/custom system prompt; Opus4.8grader; Fable5.1 safety fallback toOpus5."},{"observationId":"launch-anthropic-fable-5-1-card-11","modelId":"claude-fable-5-1","label":"Medium","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Cognition agentic coding; composite functional and code-quality score. Fable5.1 medium effort; Fable5 xhigh."},{"observationId":"launch-anthropic-fable-5-1-card-110","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Raw rubric score; adaptive max; five trials; no tools/custom system prompt; Opus4.8grader; Fable5.1 safety fallback toOpus5."},{"observationId":"launch-anthropic-fable-5-1-card-111","modelId":"claude-fable-5-1","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Length-adjusted score using GPT5.5card method; otherwise raw evaluation configuration."},{"observationId":"launch-anthropic-fable-5-1-card-112","modelId":"claude-fable-5-1","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Length-adjusted score; HealthBench Professional paper method; no tools; Opus4.8grader; five trials."},{"observationId":"launch-anthropic-fable-5-1-card-113","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Length-adjusted score; HealthBench Professional paper method; no tools; Opus4.8grader; five trials."},{"observationId":"launch-anthropic-fable-5-1-card-114","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Length-adjusted score; HealthBench Professional paper method; no tools; Opus4.8grader; five trials."},{"observationId":"launch-anthropic-fable-5-1-card-115","modelId":"claude-fable-5-1","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Mean accuracy42languages; adaptive max; one trial; no tools/custom system prompts."},{"observationId":"launch-anthropic-fable-5-1-card-116","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Mean accuracy42languages; adaptive max; one trial; no tools/custom system prompts."},{"observationId":"launch-anthropic-fable-5-1-card-117","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Mean accuracy42languages; adaptive max; one trial; no tools/custom system prompts."},{"observationId":"launch-anthropic-fable-5-1-card-118","modelId":"claude-fable-5-1","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Mean accuracy11languages; adaptive max; five trials; no tools/custom system prompts."},{"observationId":"launch-anthropic-fable-5-1-card-119","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Mean accuracy11languages; adaptive max; five trials; no tools/custom system prompts."},{"observationId":"launch-anthropic-fable-5-1-card-12","modelId":"claude-fable-5","label":"XHigh","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Cognition agentic coding; composite functional and code-quality score. Fable5.1 medium effort; Fable5 xhigh."},{"observationId":"launch-anthropic-fable-5-1-card-120","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Mean accuracy11languages; adaptive max; five trials; no tools/custom system prompts."},{"observationId":"launch-anthropic-fable-5-1-card-121","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Anthropic life-sciences evaluation; bash/editor/packages except protein design no tools; protocols use bash/editor/search; Understanding revised network restriction."},{"observationId":"launch-anthropic-fable-5-1-card-122","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Anthropic life-sciences evaluation; bash/editor/packages except protein design no tools; protocols use bash/editor/search; Understanding revised network restriction."},{"observationId":"launch-anthropic-fable-5-1-card-123","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Anthropic life-sciences evaluation; bash/editor/packages except protein design no tools; protocols use bash/editor/search; Understanding revised network restriction."},{"observationId":"launch-anthropic-fable-5-1-card-124","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Anthropic life-sciences evaluation; bash/editor/packages except protein design no tools; protocols use bash/editor/search; Understanding revised network restriction."},{"observationId":"launch-anthropic-fable-5-1-card-125","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Anthropic life-sciences evaluation; bash/editor/packages except protein design no tools; protocols use bash/editor/search; Understanding revised network restriction."},{"observationId":"launch-anthropic-fable-5-1-card-126","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Anthropic life-sciences evaluation; bash/editor/packages except protein design no tools; protocols use bash/editor/search; Understanding revised network restriction."},{"observationId":"launch-anthropic-fable-5-1-card-127","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Anthropic life-sciences evaluation; bash/editor/packages except protein design no tools; protocols use bash/editor/search; Understanding revised network restriction."},{"observationId":"launch-anthropic-fable-5-1-card-128","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Anthropic life-sciences evaluation; bash/editor/packages except protein design no tools; protocols use bash/editor/search; Understanding revised network restriction."},{"observationId":"launch-anthropic-fable-5-1-card-129","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Anthropic life-sciences evaluation; bash/editor/packages except protein design no tools; protocols use bash/editor/search; Understanding revised network restriction."},{"observationId":"launch-anthropic-fable-5-1-card-13","modelId":"claude-fable-5-1","label":"Medium","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Cognition agentic coding; composite functional and code-quality score. Fable5.1 medium effort; Fable5 xhigh."},{"observationId":"launch-anthropic-fable-5-1-card-130","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Anthropic life-sciences evaluation; bash/editor/packages except protein design no tools; protocols use bash/editor/search; Understanding revised network restriction."},{"observationId":"launch-anthropic-fable-5-1-card-131","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Anthropic life-sciences evaluation; bash/editor/packages except protein design no tools; protocols use bash/editor/search; Understanding revised network restriction."},{"observationId":"launch-anthropic-fable-5-1-card-132","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Anthropic life-sciences evaluation; bash/editor/packages except protein design no tools; protocols use bash/editor/search; Understanding revised network restriction."},{"observationId":"launch-anthropic-fable-5-1-card-133","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Anthropic life-sciences evaluation; bash/editor/packages except protein design no tools; protocols use bash/editor/search; Understanding revised network restriction."},{"observationId":"launch-anthropic-fable-5-1-card-134","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Anthropic life-sciences evaluation; bash/editor/packages except protein design no tools; protocols use bash/editor/search; Understanding revised network restriction."},{"observationId":"launch-anthropic-fable-5-1-card-135","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Anthropic life-sciences evaluation; bash/editor/packages except protein design no tools; protocols use bash/editor/search; Understanding revised network restriction."},{"observationId":"launch-anthropic-fable-5-1-card-136","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Anthropic life-sciences evaluation; bash/editor/packages except protein design no tools; protocols use bash/editor/search; Understanding revised network restriction."},{"observationId":"launch-anthropic-fable-5-1-card-14","modelId":"claude-fable-5","label":"XHigh","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Cognition agentic coding; composite functional and code-quality score. Fable5.1 medium effort; Fable5 xhigh."},{"observationId":"launch-anthropic-fable-5-1-card-15","modelId":"claude-fable-5-1","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Proximal agent harness; max effort; 34 tasks, five trials/task; mean score on 0..1 scale."},{"observationId":"launch-anthropic-fable-5-1-card-16","modelId":"claude-fable-5","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Proximal agent harness; max effort; 34 tasks, five trials/task; mean score on 0..1 scale."},{"observationId":"launch-anthropic-fable-5-1-card-17","modelId":"claude-opus-5","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Proximal agent harness; max effort; 34 tasks, five trials/task; mean score on 0..1 scale."},{"observationId":"launch-anthropic-fable-5-1-card-18","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Proximal agent harness; max effort; 34 tasks, five trials/task; mean score on 0..1 scale."},{"observationId":"launch-anthropic-fable-5-1-card-19","modelId":"claude-fable-5-1","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Claude Code --bare max effort, 15 trials/task over 66 tasks; Anthropic internal reruns."},{"observationId":"launch-anthropic-fable-5-1-card-2","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Anthropic reported configuration; adaptive thinking at max effort, default sampling, five-trial mean unless section specifies otherwise; context at most 1M. Comparator configuration follows cited prior card or board. Agent identity is not established as mini-swe-agent."},{"observationId":"launch-anthropic-fable-5-1-card-20","modelId":"claude-fable-5","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Claude Code --bare max effort, 15 trials/task over 66 tasks; Anthropic internal reruns."},{"observationId":"launch-anthropic-fable-5-1-card-21","modelId":"claude-opus-5","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Claude Code --bare max effort, 15 trials/task over 66 tasks; Anthropic internal reruns."},{"observationId":"launch-anthropic-fable-5-1-card-22","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Claude Code --bare max effort, 15 trials/task over66tasks for Claude; GPT Codex CLI max from public board."},{"observationId":"launch-anthropic-fable-5-1-card-23","modelId":"claude-fable-5-1","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"70tasks; Claude Code --bare max; Fable10trials/task, Opus12; GPT Codex CLI max from public board."},{"observationId":"launch-anthropic-fable-5-1-card-24","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"70tasks; Claude Code --bare max; Fable10trials/task, Opus12; GPT Codex CLI max from public board."},{"observationId":"launch-anthropic-fable-5-1-card-25","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"70tasks; Claude Code --bare max; Fable10trials/task, Opus12; GPT Codex CLI max from public board."},{"observationId":"launch-anthropic-fable-5-1-card-26","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"70tasks; Claude Code --bare max; Fable10trials/task, Opus12; GPT Codex CLI max from public board."},{"observationId":"launch-anthropic-fable-5-1-card-27","modelId":"claude-fable-5-1","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Cursor production agent harness; independently measured by Cursor; max effort."},{"observationId":"launch-anthropic-fable-5-1-card-28","modelId":"claude-fable-5","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Cursor production agent harness; independently measured by Cursor; max effort."},{"observationId":"launch-anthropic-fable-5-1-card-29","modelId":"claude-opus-5","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Cursor production agent harness; independently measured by Cursor; max effort."},{"observationId":"launch-anthropic-fable-5-1-card-3","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Anthropic reported configuration; adaptive thinking at max effort, default sampling, five-trial mean unless section specifies otherwise; context at most 1M. Comparator configuration follows cited prior card or board. Agent identity is not established as mini-swe-agent."},{"observationId":"launch-anthropic-fable-5-1-card-30","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Cursor production agent harness; independently measured by Cursor; max effort."},{"observationId":"launch-anthropic-fable-5-1-card-31","modelId":"claude-fable-5-1","label":"Medium","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Cursor production agent harness; medium effort; $3.53/task."},{"observationId":"launch-anthropic-fable-5-1-card-32","modelId":"claude-fable-5-1","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"mini-swe-agent without six-hour timeout; excludes34flaky-reference tasks; tests restricted to reference-passing tests; up to1Mcontext."},{"observationId":"launch-anthropic-fable-5-1-card-33","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"mini-swe-agent without six-hour timeout; excludes34flaky-reference tasks; tests restricted to reference-passing tests; up to1Mcontext."},{"observationId":"launch-anthropic-fable-5-1-card-34","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"mini-swe-agent without six-hour timeout; excludes34flaky-reference tasks; tests restricted to reference-passing tests; up to1Mcontext."},{"observationId":"launch-anthropic-fable-5-1-card-35","modelId":"claude-fable-5-1","label":"Auto thinking","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Full2500questions; no tools; auto thinking;1Mtotal token cap; no compaction; Opus4.6grader; restricted fetch and contamination review for tools."},{"observationId":"launch-anthropic-fable-5-1-card-36","modelId":"claude-fable-5","label":"Auto thinking","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Full2500questions; no tools; auto thinking;1Mtotal token cap; no compaction; Opus4.6grader; restricted fetch and contamination review for tools."},{"observationId":"launch-anthropic-fable-5-1-card-37","modelId":"claude-opus-5","label":"Auto thinking","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Full2500questions; no tools; auto thinking;1Mtotal token cap; no compaction; Opus4.6grader; restricted fetch and contamination review for tools."},{"observationId":"launch-anthropic-fable-5-1-card-38","modelId":"claude-fable-5-1","label":"Auto thinking","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Full2500questions; with tools; auto thinking;1Mtotal token cap; no compaction; Opus4.6grader; restricted fetch and contamination review for tools."},{"observationId":"launch-anthropic-fable-5-1-card-39","modelId":"claude-fable-5","label":"Auto thinking","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Full2500questions; with tools; auto thinking;1Mtotal token cap; no compaction; Opus4.6grader; restricted fetch and contamination review for tools."},{"observationId":"launch-anthropic-fable-5-1-card-4","modelId":"claude-fable-5-1","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Anthropic reported configuration; adaptive thinking at max effort, default sampling, five-trial mean unless section specifies otherwise; context at most 1M. Comparator configuration follows cited prior card or board. Agent identity is not established as mini-swe-agent."},{"observationId":"launch-anthropic-fable-5-1-card-40","modelId":"claude-opus-5","label":"Auto thinking","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Full2500questions; with tools; auto thinking;1Mtotal token cap; no compaction; Opus4.6grader; restricted fetch and contamination review for tools."},{"observationId":"launch-anthropic-fable-5-1-card-41","modelId":"claude-fable-5-1","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"100tasks; adaptive thinking max; five runs; no tools; tools condition has container, standard libraries and crop tool."},{"observationId":"launch-anthropic-fable-5-1-card-42","modelId":"claude-fable-5","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"100tasks; adaptive thinking max; five runs; no tools; tools condition has container, standard libraries and crop tool."},{"observationId":"launch-anthropic-fable-5-1-card-43","modelId":"claude-opus-5","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"100tasks; adaptive thinking max; five runs; no tools; tools condition has container, standard libraries and crop tool."},{"observationId":"launch-anthropic-fable-5-1-card-44","modelId":"claude-fable-5-1","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"100tasks; adaptive thinking max; five runs; with tools; tools condition has container, standard libraries and crop tool."},{"observationId":"launch-anthropic-fable-5-1-card-45","modelId":"claude-fable-5","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"100tasks; adaptive thinking max; five runs; with tools; tools condition has container, standard libraries and crop tool."},{"observationId":"launch-anthropic-fable-5-1-card-46","modelId":"claude-opus-5","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"100tasks; adaptive thinking max; five runs; with tools; tools condition has container, standard libraries and crop tool."},{"observationId":"launch-anthropic-fable-5-1-card-47","modelId":"claude-fable-5-1","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Random1000 of17900files; five runs; adaptive thinking max; no tools; corrected camera prompt, raw shapes accepted, last code fence parsed."},{"observationId":"launch-anthropic-fable-5-1-card-48","modelId":"claude-fable-5","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Random1000 of17900files; five runs; adaptive thinking max; no tools; corrected camera prompt, raw shapes accepted, last code fence parsed."},{"observationId":"launch-anthropic-fable-5-1-card-49","modelId":"claude-opus-5","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Random1000 of17900files; five runs; adaptive thinking max; no tools; corrected camera prompt, raw shapes accepted, last code fence parsed."},{"observationId":"launch-anthropic-fable-5-1-card-5","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Anthropic reported configuration; adaptive thinking at max effort, default sampling, five-trial mean unless section specifies otherwise; context at most 1M. Comparator configuration follows cited prior card or board. Agent identity is not established as mini-swe-agent."},{"observationId":"launch-anthropic-fable-5-1-card-50","modelId":"claude-fable-5-1","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Random1000 of17900files; five runs; adaptive thinking max; with tools; corrected camera prompt, raw shapes accepted, last code fence parsed."},{"observationId":"launch-anthropic-fable-5-1-card-51","modelId":"claude-fable-5","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Random1000 of17900files; five runs; adaptive thinking max; with tools; corrected camera prompt, raw shapes accepted, last code fence parsed."},{"observationId":"launch-anthropic-fable-5-1-card-52","modelId":"claude-opus-5","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Random1000 of17900files; five runs; adaptive thinking max; with tools; corrected camera prompt, raw shapes accepted, last code fence parsed."},{"observationId":"launch-anthropic-fable-5-1-card-53","modelId":"claude-fable-5-1","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"partial pass@1;108tasks; five runs;1080p;500steps; max effort; Opus4.8grader; task fixes; Fable safety interventions score zero."},{"observationId":"launch-anthropic-fable-5-1-card-54","modelId":"claude-fable-5","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"partial pass@1;108tasks; five runs;1080p;500steps; max effort; Opus4.8grader; task fixes; Fable safety interventions score zero."},{"observationId":"launch-anthropic-fable-5-1-card-55","modelId":"claude-opus-5","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"partial pass@1;108tasks; five runs;1080p;500steps; max effort; Opus4.8grader; task fixes; Fable safety interventions score zero."},{"observationId":"launch-anthropic-fable-5-1-card-56","modelId":"claude-fable-5-1","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"strict pass@1;108tasks; five runs;1080p;500steps; max effort; Opus4.8grader; task fixes; Fable safety interventions score zero."},{"observationId":"launch-anthropic-fable-5-1-card-57","modelId":"claude-fable-5","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"strict pass@1;108tasks; five runs;1080p;500steps; max effort; Opus4.8grader; task fixes; Fable safety interventions score zero."},{"observationId":"launch-anthropic-fable-5-1-card-58","modelId":"claude-opus-5","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"strict pass@1;108tasks; five runs;1080p;500steps; max effort; Opus4.8grader; task fixes; Fable safety interventions score zero."},{"observationId":"launch-anthropic-fable-5-1-card-59","modelId":"claude-fable-5-1","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Extracted-text Treasury corpus in sandbox; code execution; production Messages API with safeguards/fallback;128koutput cap. Pro133questions."},{"observationId":"launch-anthropic-fable-5-1-card-6","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Anthropic reported configuration; adaptive thinking at max effort, default sampling, five-trial mean unless section specifies otherwise; context at most 1M. Comparator configuration follows cited prior card or board. Agent identity is not established as mini-swe-agent."},{"observationId":"launch-anthropic-fable-5-1-card-60","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Extracted-text Treasury corpus in sandbox; code execution; production Messages API with safeguards/fallback;128koutput cap. Pro133questions."},{"observationId":"launch-anthropic-fable-5-1-card-61","modelId":"claude-fable-5-1","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Extracted-text Treasury corpus in sandbox; code execution; production Messages API with safeguards/fallback;128koutput cap. Pro133questions."},{"observationId":"launch-anthropic-fable-5-1-card-62","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Extracted-text Treasury corpus in sandbox; code execution; production Messages API with safeguards/fallback;128koutput cap. Pro133questions."},{"observationId":"launch-anthropic-fable-5-1-card-63","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Databricks evaluation reading documents as images; differs from extracted-text harness."},{"observationId":"launch-anthropic-fable-5-1-card-64","modelId":"claude-fable-5-1","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"all-pass; five runs; adaptive max; internal bash/Python harness, Sonnet4.6judge;16defective tasks excluded; production safeguards/fallback."},{"observationId":"launch-anthropic-fable-5-1-card-65","modelId":"claude-fable-5-1","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"criterion-pass; five runs; adaptive max; internal bash/Python harness, Sonnet4.6judge;16defective tasks excluded; production safeguards/fallback."},{"observationId":"launch-anthropic-fable-5-1-card-66","modelId":"claude-fable-5-1","label":"XHigh","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"all-pass; Artificial Analysis harness; xhigh effort."},{"observationId":"launch-anthropic-fable-5-1-card-67","modelId":"claude-fable-5-1","label":"XHigh","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"criterion-pass; Artificial Analysis harness; xhigh effort."},{"observationId":"launch-anthropic-fable-5-1-card-68","modelId":"claude-fable-5-1","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Artificial Analysis independent agentic shell/web evaluation;220GDPvalgoldtasks; blind pairwise Elo; max effort for Claude."},{"observationId":"launch-anthropic-fable-5-1-card-69","modelId":"claude-fable-5","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Artificial Analysis independent agentic shell/web evaluation;220GDPvalgoldtasks; blind pairwise Elo; max effort for Claude."},{"observationId":"launch-anthropic-fable-5-1-card-7","modelId":"claude-fable-5-1","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Anthropic reported configuration; adaptive thinking at max effort, default sampling, five-trial mean unless section specifies otherwise; context at most 1M. Comparator configuration follows cited prior card or board. Agent identity is not established as mini-swe-agent."},{"observationId":"launch-anthropic-fable-5-1-card-70","modelId":"claude-opus-5","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Artificial Analysis independent agentic shell/web evaluation;220GDPvalgoldtasks; blind pairwise Elo; max effort for Claude."},{"observationId":"launch-anthropic-fable-5-1-card-71","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Artificial Analysis independent agentic shell/web evaluation;220GDPvalgoldtasks; blind pairwise Elo; max effort for Claude."},{"observationId":"launch-anthropic-fable-5-1-card-72","modelId":"claude-fable-5-1","label":"XHigh","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Artificial Analysis; xhigh effort; same release board snapshot."},{"observationId":"launch-anthropic-fable-5-1-card-73","modelId":"claude-fable-5-1","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Artificial Analysis long-horizon knowledge projects; rubric and panel pairwise judging; Claude max effort."},{"observationId":"launch-anthropic-fable-5-1-card-74","modelId":"claude-fable-5","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Artificial Analysis long-horizon knowledge projects; rubric and panel pairwise judging; Claude max effort."},{"observationId":"launch-anthropic-fable-5-1-card-75","modelId":"claude-opus-5","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Artificial Analysis long-horizon knowledge projects; rubric and panel pairwise judging; Claude max effort."},{"observationId":"launch-anthropic-fable-5-1-card-76","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Artificial Analysis long-horizon knowledge projects; rubric and panel pairwise judging; Claude max effort."},{"observationId":"launch-anthropic-fable-5-1-card-77","modelId":"claude-fable-5-1","label":"XHigh","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Artificial Analysis; xhigh effort."},{"observationId":"launch-anthropic-fable-5-1-card-78","modelId":"claude-fable-5-1","label":"High","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Artificial Analysis; high effort."},{"observationId":"launch-anthropic-fable-5-1-card-79","modelId":"claude-fable-5-1","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Artificial Analysis; max effort; component rubric pass."},{"observationId":"launch-anthropic-fable-5-1-card-8","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Anthropic reported configuration; adaptive thinking at max effort, default sampling, five-trial mean unless section specifies otherwise; context at most 1M. Comparator configuration follows cited prior card or board. Agent identity is not established as mini-swe-agent."},{"observationId":"launch-anthropic-fable-5-1-card-80","modelId":"claude-opus-5","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Artificial Analysis; max effort; component rubric pass."},{"observationId":"launch-anthropic-fable-5-1-card-81","modelId":"claude-fable-5-1","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Artificial Analysis; max effort; component analytical quality."},{"observationId":"launch-anthropic-fable-5-1-card-82","modelId":"claude-opus-5","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Artificial Analysis; max effort; component analytical quality."},{"observationId":"launch-anthropic-fable-5-1-card-83","modelId":"claude-fable-5-1","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Artificial Analysis; max effort; component presentation."},{"observationId":"launch-anthropic-fable-5-1-card-84","modelId":"claude-opus-5","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Artificial Analysis; max effort; component presentation."},{"observationId":"launch-anthropic-fable-5-1-card-85","modelId":"claude-fable-5-1","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Pass@1;108tasks/three trials; internal harness; max effort; Fable safeguards+Opus4.8fallback; Opus safeguards/fallback disabled; pinned containers/data."},{"observationId":"launch-anthropic-fable-5-1-card-86","modelId":"claude-opus-5","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Pass@1;108tasks/three trials; internal harness; max effort; Fable safeguards+Opus4.8fallback; Opus safeguards/fallback disabled; pinned containers/data."},{"observationId":"launch-anthropic-fable-5-1-card-87","modelId":"claude-fable-5-1","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Pass@3;108tasks/three trials; internal harness; max effort; Fable safeguards+Opus4.8fallback; Opus safeguards/fallback disabled; pinned containers/data."},{"observationId":"launch-anthropic-fable-5-1-card-88","modelId":"claude-opus-5","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Pass@3;108tasks/three trials; internal harness; max effort; Fable safeguards+Opus4.8fallback; Opus safeguards/fallback disabled; pinned containers/data."},{"observationId":"launch-anthropic-fable-5-1-card-89","modelId":"claude-fable-5-1","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Pass³;108tasks/three trials; internal harness; max effort; Fable safeguards+Opus4.8fallback; Opus safeguards/fallback disabled; pinned containers/data."},{"observationId":"launch-anthropic-fable-5-1-card-9","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Anthropic reported configuration; adaptive thinking at max effort, default sampling, five-trial mean unless section specifies otherwise; context at most 1M. Comparator configuration follows cited prior card or board. Agent identity is not established as mini-swe-agent."},{"observationId":"launch-anthropic-fable-5-1-card-90","modelId":"claude-opus-5","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Pass³;108tasks/three trials; internal harness; max effort; Fable safeguards+Opus4.8fallback; Opus safeguards/fallback disabled; pinned containers/data."},{"observationId":"launch-anthropic-fable-5-1-card-91","modelId":"claude-fable-5-1","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"average turns;108tasks/three trials; internal harness; max effort; Fable safeguards+Opus4.8fallback; Opus safeguards/fallback disabled; pinned containers/data."},{"observationId":"launch-anthropic-fable-5-1-card-92","modelId":"claude-opus-5","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"average turns;108tasks/three trials; internal harness; max effort; Fable safeguards+Opus4.8fallback; Opus safeguards/fallback disabled; pinned containers/data."},{"observationId":"launch-anthropic-fable-5-1-card-93","modelId":"claude-fable-5-1","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Private held-out board; simulated business-workflow app APIs; all assertions must pass; max effort stated for Fable5.1/Opus5."},{"observationId":"launch-anthropic-fable-5-1-card-94","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Private held-out board; simulated business-workflow app APIs; all assertions must pass; max effort stated for Fable5.1/Opus5."},{"observationId":"launch-anthropic-fable-5-1-card-95","modelId":"claude-opus-5","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Private held-out board; simulated business-workflow app APIs; all assertions must pass; max effort stated for Fable5.1/Opus5."},{"observationId":"launch-anthropic-fable-5-1-card-96","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"Private held-out board; simulated business-workflow app APIs; all assertions must pass; max effort stated for Fable5.1/Opus5."},{"observationId":"launch-anthropic-fable-5-1-card-97","modelId":"claude-fable-5-1","label":"Max","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"ARC Prize semi-private validation; verified Fable5.1 max effort; comparator figures as reported in summary."},{"observationId":"launch-anthropic-fable-5-1-card-98","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"ARC Prize semi-private validation; verified Fable5.1 max effort; comparator figures as reported in summary."},{"observationId":"launch-anthropic-fable-5-1-card-99","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":"ARC Prize semi-private validation; verified Fable5.1 max effort; comparator figures as reported in summary."},{"observationId":"launch-cohere-1429","modelId":"command-a-plus-05-2026","label":"Not specified","sourceUrl":"https://cohere.com/blog/command-a-plus","configuration":"Cohere launch evaluation; North metric uses internal LLM judge."},{"observationId":"launch-cohere-1430","modelId":"command-a-reasoning-08-2025","label":"Not specified","sourceUrl":"https://cohere.com/blog/command-a-plus","configuration":"Cohere launch evaluation; North metric uses internal LLM judge."},{"observationId":"launch-cohere-1431","modelId":"command-a-plus-05-2026","label":"Not specified","sourceUrl":"https://cohere.com/blog/command-a-plus","configuration":"Cohere launch evaluation; North metric uses internal LLM judge."},{"observationId":"launch-cohere-1432","modelId":"command-a-reasoning-08-2025","label":"Not specified","sourceUrl":"https://cohere.com/blog/command-a-plus","configuration":"Cohere launch evaluation; North metric uses internal LLM judge."},{"observationId":"launch-cohere-1433","modelId":"command-a-plus-05-2026","label":"Not specified","sourceUrl":"https://cohere.com/blog/command-a-plus","configuration":"Cohere launch evaluation; North metric uses internal LLM judge."},{"observationId":"launch-cohere-1434","modelId":"command-a-reasoning-08-2025","label":"Not specified","sourceUrl":"https://cohere.com/blog/command-a-plus","configuration":"Cohere launch evaluation; North metric uses internal LLM judge."},{"observationId":"launch-cohere-1435","modelId":"command-a-plus-05-2026","label":"Not specified","sourceUrl":"https://cohere.com/blog/command-a-plus","configuration":"Cohere launch multimodal evaluations."},{"observationId":"launch-cohere-1436","modelId":"command-a-plus-05-2026","label":"Not specified","sourceUrl":"https://cohere.com/blog/command-a-plus","configuration":"Cohere launch multimodal evaluations."},{"observationId":"launch-cohere-1437","modelId":"command-a-vision-07-2025","label":"Not specified","sourceUrl":"https://cohere.com/blog/command-a-plus","configuration":"Cohere launch multimodal evaluations."},{"observationId":"launch-cohere-1438","modelId":"command-a-plus-05-2026","label":"Not specified","sourceUrl":"https://cohere.com/blog/command-a-plus","configuration":"Cohere launch multimodal evaluations."},{"observationId":"launch-cohere-1439","modelId":"command-a-vision-07-2025","label":"Not specified","sourceUrl":"https://cohere.com/blog/command-a-plus","configuration":"Cohere launch multimodal evaluations."},{"observationId":"launch-cohere-1440","modelId":"command-a-plus-05-2026","label":"Not specified","sourceUrl":"https://cohere.com/blog/command-a-plus","configuration":"Cohere launch multimodal evaluations."},{"observationId":"launch-cohere-1441","modelId":"command-a-vision-07-2025","label":"Not specified","sourceUrl":"https://cohere.com/blog/command-a-plus","configuration":"Cohere launch multimodal evaluations."},{"observationId":"launch-cohere-1967","modelId":"command-a-plus-05-2026","label":"Not specified","sourceUrl":"https://cohere.com/blog/command-a-plus","configuration":"Single-turn loose,prompt accuracy,294prompts x5 repeats."},{"observationId":"launch-cohere-1968","modelId":"command-a-plus-05-2026","label":"Not specified","sourceUrl":"https://cohere.com/blog/command-a-plus","configuration":"Official30questions x10 repeats; pass@1."},{"observationId":"launch-cohere-1969","modelId":"command-a-plus-05-2026","label":"Not specified","sourceUrl":"https://cohere.com/blog/command-a-plus","configuration":"65problems/288subproblems; scientist-annotated background."},{"observationId":"launch-cohere-1970","modelId":"command-a-plus-05-2026","label":"Not specified","sourceUrl":"https://cohere.com/blog/command-a-plus","configuration":"Internal North enterprise MCP cloud-file QA,LLM judge."},{"observationId":"launch-cohere-1971","modelId":"command-a-plus-05-2026","label":"Not specified","sourceUrl":"https://cohere.com/blog/command-a-plus","configuration":"Internal North uploaded spreadsheet data-science tasks,LLM judge."},{"observationId":"launch-cohere-1972","modelId":"command-a-plus-05-2026","label":"Not specified","sourceUrl":"https://cohere.com/blog/command-a-plus","configuration":"Internal Command A Translate translations; Arabic,Japanese,Korean."},{"observationId":"launch-cohere-1973","modelId":"command-a-plus-05-2026","label":"Not specified","sourceUrl":"https://cohere.com/blog/command-a-plus","configuration":"xCOMETxl average50 varieties,including internal Irish/Maltese translations and Serbian transliteration."},{"observationId":"launch-cohere-1974","modelId":"command-a-reasoning-08-2025","label":"Not specified","sourceUrl":"https://cohere.com/blog/command-a-plus","configuration":"Single-turn loose,prompt accuracy,294prompts x5 repeats."},{"observationId":"launch-cohere-1975","modelId":"command-a-reasoning-08-2025","label":"Not specified","sourceUrl":"https://cohere.com/blog/command-a-plus","configuration":"Official30questions x10 repeats; pass@1."},{"observationId":"launch-cohere-1976","modelId":"command-a-reasoning-08-2025","label":"Not specified","sourceUrl":"https://cohere.com/blog/command-a-plus","configuration":"65problems/288subproblems; scientist-annotated background."},{"observationId":"launch-cohere-1977","modelId":"command-a-reasoning-08-2025","label":"Not specified","sourceUrl":"https://cohere.com/blog/command-a-plus","configuration":"Internal North enterprise MCP cloud-file QA,LLM judge."},{"observationId":"launch-cohere-1978","modelId":"command-a-reasoning-08-2025","label":"Not specified","sourceUrl":"https://cohere.com/blog/command-a-plus","configuration":"Internal North uploaded spreadsheet data-science tasks,LLM judge."},{"observationId":"launch-cohere-1979","modelId":"command-a-reasoning-08-2025","label":"Not specified","sourceUrl":"https://cohere.com/blog/command-a-plus","configuration":"Internal Command A Translate translations; Arabic,Japanese,Korean."},{"observationId":"launch-cohere-1980","modelId":"command-a-reasoning-08-2025","label":"Not specified","sourceUrl":"https://cohere.com/blog/command-a-plus","configuration":"xCOMETxl average50 varieties,including internal Irish/Maltese translations and Serbian transliteration."},{"observationId":"launch-cohere-1981","modelId":"command-a-plus-05-2026","label":"Not specified","sourceUrl":"https://cohere.com/blog/command-a-plus","configuration":"Standard methodology; integer rounded labels in chart."},{"observationId":"launch-cohere-1982","modelId":"command-a-vision-07-2025","label":"Not specified","sourceUrl":"https://cohere.com/blog/command-a-plus","configuration":"Standard methodology; integer rounded labels in chart."},{"observationId":"launch-cohere-1983","modelId":"command-a-plus-05-2026","label":"Not specified","sourceUrl":"https://cohere.com/blog/command-a-plus","configuration":"Cohere quoted AA launch snapshot; index version unspecified."},{"observationId":"launch-deepseek-1136","modelId":"deepseek-v4-pro","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card. without tools"},{"observationId":"launch-deepseek-1137","modelId":"deepseek-v4-pro","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card. with tools"},{"observationId":"launch-deepseek-1138","modelId":"deepseek-v4-flash","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card. without tools"},{"observationId":"launch-deepseek-1139","modelId":"deepseek-v4-flash","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card. with tools"},{"observationId":"launch-deepseek-1140","modelId":"glm-5.2","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card. without tools"},{"observationId":"launch-deepseek-1141","modelId":"glm-5.2","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card. with tools"},{"observationId":"launch-deepseek-1142","modelId":"kimi-k3","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card. without tools"},{"observationId":"launch-deepseek-1143","modelId":"kimi-k3","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card. with tools"},{"observationId":"launch-deepseek-1144","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card. without tools"},{"observationId":"launch-deepseek-1145","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card. with tools"},{"observationId":"launch-deepseek-1146","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card. without tools"},{"observationId":"launch-deepseek-1147","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card. with tools"},{"observationId":"launch-deepseek-1148","modelId":"deepseek-v4-pro","label":"Max","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"observationId":"launch-deepseek-1149","modelId":"deepseek-v4-flash","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"observationId":"launch-deepseek-1150","modelId":"glm-5.2","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"observationId":"launch-deepseek-1151","modelId":"kimi-k3","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"observationId":"launch-deepseek-1152","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"observationId":"launch-deepseek-1153","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"observationId":"launch-deepseek-1154","modelId":"deepseek-v4-pro","label":"Max","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"observationId":"launch-deepseek-1155","modelId":"deepseek-v4-flash","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"observationId":"launch-deepseek-1156","modelId":"glm-5.2","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"observationId":"launch-deepseek-1157","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"observationId":"launch-deepseek-1158","modelId":"deepseek-v4-pro","label":"Max","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"observationId":"launch-deepseek-1159","modelId":"deepseek-v4-flash","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"observationId":"launch-deepseek-1160","modelId":"kimi-k3","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"observationId":"launch-deepseek-1161","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"observationId":"launch-deepseek-1162","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"observationId":"launch-deepseek-1163","modelId":"deepseek-v4-pro","label":"Max","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"observationId":"launch-deepseek-1164","modelId":"deepseek-v4-flash","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"observationId":"launch-deepseek-1165","modelId":"glm-5.2","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"observationId":"launch-deepseek-1166","modelId":"kimi-k3","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"observationId":"launch-deepseek-1167","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"observationId":"launch-deepseek-1168","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"observationId":"launch-deepseek-1169","modelId":"deepseek-v4-pro","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"observationId":"launch-deepseek-1170","modelId":"deepseek-v4-flash","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"observationId":"launch-deepseek-1171","modelId":"glm-5.2","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"observationId":"launch-deepseek-1172","modelId":"kimi-k3","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"observationId":"launch-deepseek-1173","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"observationId":"launch-deepseek-1174","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"observationId":"launch-deepseek-1175","modelId":"deepseek-v4-pro","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"observationId":"launch-deepseek-1176","modelId":"deepseek-v4-flash","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"observationId":"launch-deepseek-1177","modelId":"glm-5.2","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"observationId":"launch-deepseek-1178","modelId":"kimi-k3","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"observationId":"launch-deepseek-1179","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"observationId":"launch-deepseek-1180","modelId":"deepseek-v4-pro","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"observationId":"launch-deepseek-1181","modelId":"deepseek-v4-flash","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"observationId":"launch-deepseek-1182","modelId":"glm-5.2","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"observationId":"launch-deepseek-1183","modelId":"kimi-k3","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"observationId":"launch-deepseek-1184","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"observationId":"launch-deepseek-1185","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"observationId":"launch-deepseek-1186","modelId":"deepseek-v4-pro","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"observationId":"launch-deepseek-1187","modelId":"deepseek-v4-flash","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"observationId":"launch-deepseek-1188","modelId":"glm-5.2","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"observationId":"launch-deepseek-1189","modelId":"kimi-k3","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"observationId":"launch-deepseek-1190","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"observationId":"launch-deepseek-1191","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"observationId":"launch-deepseek-1192","modelId":"deepseek-v4-pro","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"observationId":"launch-deepseek-1193","modelId":"deepseek-v4-flash","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"observationId":"launch-deepseek-1194","modelId":"glm-5.2","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"observationId":"launch-deepseek-1195","modelId":"kimi-k3","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"observationId":"launch-deepseek-1196","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"observationId":"launch-deepseek-1197","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/README.md","configuration":"DeepSeek updated 0813/0731 release evaluation table; tool/harness settings per model card."},{"observationId":"launch-google-1357","modelId":"gemini-3.8-flash","label":"Not specified","sourceUrl":"https://deepmind.google/models/gemini/flash/","configuration":"Google launch chart; benchmark methodology linked on page; reported comparator settings vary."},{"observationId":"launch-google-1358","modelId":"gemini-3.7-flash","label":"Not specified","sourceUrl":"https://deepmind.google/models/gemini/flash/","configuration":"Google launch chart; benchmark methodology linked on page; reported comparator settings vary."},{"observationId":"launch-google-1359","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://deepmind.google/models/gemini/flash/","configuration":"Google launch chart; benchmark methodology linked on page; reported comparator settings vary."},{"observationId":"launch-google-1360","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://deepmind.google/models/gemini/flash/","configuration":"Google launch chart; benchmark methodology linked on page; reported comparator settings vary."},{"observationId":"launch-google-1361","modelId":"claude-sonnet-5","label":"Not specified","sourceUrl":"https://deepmind.google/models/gemini/flash/","configuration":"Google launch chart; benchmark methodology linked on page; reported comparator settings vary."},{"observationId":"launch-google-1362","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://deepmind.google/models/gemini/flash/","configuration":"Google launch chart; benchmark methodology linked on page; reported comparator settings vary."},{"observationId":"launch-google-1363","modelId":"gemini-3.8-flash","label":"Not specified","sourceUrl":"https://deepmind.google/models/gemini/flash/","configuration":"Google launch chart; benchmark methodology linked on page; reported comparator settings vary."},{"observationId":"launch-google-1364","modelId":"gemini-3.7-flash","label":"Not specified","sourceUrl":"https://deepmind.google/models/gemini/flash/","configuration":"Google launch chart; benchmark methodology linked on page; reported comparator settings vary."},{"observationId":"launch-google-1365","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://deepmind.google/models/gemini/flash/","configuration":"Google launch chart; benchmark methodology linked on page; reported comparator settings vary."},{"observationId":"launch-google-1366","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://deepmind.google/models/gemini/flash/","configuration":"Google launch chart; benchmark methodology linked on page; reported comparator settings vary."},{"observationId":"launch-google-1367","modelId":"claude-sonnet-5","label":"Not specified","sourceUrl":"https://deepmind.google/models/gemini/flash/","configuration":"Google launch chart; benchmark methodology linked on page; reported comparator settings vary."},{"observationId":"launch-google-1368","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://deepmind.google/models/gemini/flash/","configuration":"Google launch chart; benchmark methodology linked on page; reported comparator settings vary."},{"observationId":"launch-google-1369","modelId":"gemini-3.8-flash","label":"Not specified","sourceUrl":"https://deepmind.google/models/gemini/flash/","configuration":"Google launch chart; benchmark methodology linked on page; reported comparator settings vary."},{"observationId":"launch-google-1370","modelId":"gemini-3.7-flash","label":"Not specified","sourceUrl":"https://deepmind.google/models/gemini/flash/","configuration":"Google launch chart; benchmark methodology linked on page; reported comparator settings vary."},{"observationId":"launch-google-1371","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://deepmind.google/models/gemini/flash/","configuration":"Google launch chart; benchmark methodology linked on page; reported comparator settings vary."},{"observationId":"launch-google-1372","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://deepmind.google/models/gemini/flash/","configuration":"Google launch chart; benchmark methodology linked on page; reported comparator settings vary."},{"observationId":"launch-google-1373","modelId":"claude-sonnet-5","label":"Not specified","sourceUrl":"https://deepmind.google/models/gemini/flash/","configuration":"Google launch chart; benchmark methodology linked on page; reported comparator settings vary."},{"observationId":"launch-google-1374","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://deepmind.google/models/gemini/flash/","configuration":"Google launch chart; benchmark methodology linked on page; reported comparator settings vary."},{"observationId":"launch-google-gemini-3-8-card-2040","modelId":"gemini-3.8-flash","label":"High thinking","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"Datacurve highest reported effort; Gemini3.8selfcomputed mini-swe-agent highthinking"},{"observationId":"launch-google-gemini-3-8-card-2041","modelId":"gemini-3.7-flash","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"Datacurve highest reported effort; Gemini3.8selfcomputed mini-swe-agent highthinking"},{"observationId":"launch-google-gemini-3-8-card-2042","modelId":"claude-sonnet-5","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"Datacurve highest reported effort; Gemini3.8selfcomputed mini-swe-agent highthinking"},{"observationId":"launch-google-gemini-3-8-card-2043","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"Datacurve highest reported effort; Gemini3.8selfcomputed mini-swe-agent highthinking"},{"observationId":"launch-google-gemini-3-8-card-2044","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"Datacurve highest reported effort; Gemini3.8selfcomputed mini-swe-agent highthinking"},{"observationId":"launch-google-gemini-3-8-card-2045","modelId":"gemini-3.8-flash","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"Artificial Analysis publicboard snapshot; effort as reported"},{"observationId":"launch-google-gemini-3-8-card-2046","modelId":"gemini-3.7-flash","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"Artificial Analysis publicboard snapshot; effort as reported"},{"observationId":"launch-google-gemini-3-8-card-2047","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"Artificial Analysis publicboard snapshot; effort as reported"},{"observationId":"launch-google-gemini-3-8-card-2048","modelId":"claude-sonnet-5","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"Artificial Analysis publicboard snapshot; effort as reported"},{"observationId":"launch-google-gemini-3-8-card-2049","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"Artificial Analysis publicboard snapshot; effort as reported"},{"observationId":"launch-google-gemini-3-8-card-2050","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"Artificial Analysis publicboard snapshot; effort as reported"},{"observationId":"launch-google-gemini-3-8-card-2051","modelId":"gemini-3.8-flash","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"Terminus2only; Gemini selfcomputed, othermodels officialboard/ArtificialAnalysis"},{"observationId":"launch-google-gemini-3-8-card-2052","modelId":"gemini-3.7-flash","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"Terminus2only; Gemini selfcomputed, othermodels officialboard/ArtificialAnalysis"},{"observationId":"launch-google-gemini-3-8-card-2053","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"Terminus2only; Gemini selfcomputed, othermodels officialboard/ArtificialAnalysis"},{"observationId":"launch-google-gemini-3-8-card-2054","modelId":"claude-sonnet-5","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"Terminus2only; Gemini selfcomputed, othermodels officialboard/ArtificialAnalysis"},{"observationId":"launch-google-gemini-3-8-card-2055","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"Terminus2only; Gemini selfcomputed, othermodels officialboard/ArtificialAnalysis"},{"observationId":"launch-google-gemini-3-8-card-2056","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"Terminus2only; Gemini selfcomputed, othermodels officialboard/ArtificialAnalysis"},{"observationId":"launch-google-gemini-3-8-card-2057","modelId":"gemini-3.8-flash","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"Officialpublicboard highest scoring thinking level; nativeagents may differ"},{"observationId":"launch-google-gemini-3-8-card-2058","modelId":"gemini-3.7-flash","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"Officialpublicboard highest scoring thinking level; nativeagents may differ"},{"observationId":"launch-google-gemini-3-8-card-2059","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"Officialpublicboard highest scoring thinking level; nativeagents may differ"},{"observationId":"launch-google-gemini-3-8-card-2060","modelId":"claude-sonnet-5","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"Officialpublicboard highest scoring thinking level; nativeagents may differ"},{"observationId":"launch-google-gemini-3-8-card-2061","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"Officialpublicboard highest scoring thinking level; nativeagents may differ"},{"observationId":"launch-google-gemini-3-8-card-2062","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"Officialpublicboard highest scoring thinking level; nativeagents may differ"},{"observationId":"launch-google-gemini-3-8-card-2063","modelId":"gemini-3.8-flash","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"All-pass rate; allmodels selfcomputed byGoogle"},{"observationId":"launch-google-gemini-3-8-card-2064","modelId":"gemini-3.7-flash","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"All-pass rate; allmodels selfcomputed byGoogle"},{"observationId":"launch-google-gemini-3-8-card-2065","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"All-pass rate; allmodels selfcomputed byGoogle"},{"observationId":"launch-google-gemini-3-8-card-2066","modelId":"claude-sonnet-5","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"All-pass rate; allmodels selfcomputed byGoogle"},{"observationId":"launch-google-gemini-3-8-card-2067","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"All-pass rate; allmodels selfcomputed byGoogle"},{"observationId":"launch-google-gemini-3-8-card-2068","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"All-pass rate; allmodels selfcomputed byGoogle"},{"observationId":"launch-google-gemini-3-8-card-2069","modelId":"gemini-3.8-flash","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"No tools; Gemini/GPT/Opus selfcomputed; Sonnet selfreported"},{"observationId":"launch-google-gemini-3-8-card-2070","modelId":"gemini-3.7-flash","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"No tools; Gemini/GPT/Opus selfcomputed; Sonnet selfreported"},{"observationId":"launch-google-gemini-3-8-card-2071","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"No tools; Gemini/GPT/Opus selfcomputed; Sonnet selfreported"},{"observationId":"launch-google-gemini-3-8-card-2072","modelId":"claude-sonnet-5","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"No tools; Gemini/GPT/Opus selfcomputed; Sonnet selfreported"},{"observationId":"launch-google-gemini-3-8-card-2073","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"No tools; Gemini/GPT/Opus selfcomputed; Sonnet selfreported"},{"observationId":"launch-google-gemini-3-8-card-2074","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"No tools; Gemini/GPT/Opus selfcomputed; Sonnet selfreported"},{"observationId":"launch-google-gemini-3-8-card-2075","modelId":"gemini-3.8-flash","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"No tools;1024frames Gemini/GPT,300frames Claude dueAPIlimit; model-specific frame budget: 1024"},{"observationId":"launch-google-gemini-3-8-card-2076","modelId":"gemini-3.7-flash","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"No tools;1024frames Gemini/GPT,300frames Claude dueAPIlimit; model-specific frame budget: 1024"},{"observationId":"launch-google-gemini-3-8-card-2077","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"No tools;1024frames Gemini/GPT,300frames Claude dueAPIlimit; model-specific frame budget: 300"},{"observationId":"launch-google-gemini-3-8-card-2078","modelId":"claude-sonnet-5","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"No tools;1024frames Gemini/GPT,300frames Claude dueAPIlimit; model-specific frame budget: 300"},{"observationId":"launch-google-gemini-3-8-card-2079","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"No tools;1024frames Gemini/GPT,300frames Claude dueAPIlimit; model-specific frame budget: 1024"},{"observationId":"launch-google-gemini-3-8-card-2080","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"No tools;1024frames Gemini/GPT,300frames Claude dueAPIlimit; model-specific frame budget: 1024"},{"observationId":"launch-google-gemini-3-8-card-2081","modelId":"gemini-3.8-flash","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"Card labels agentic; linkedmethodology describes only no-tools static setup"},{"observationId":"launch-google-gemini-3-8-card-2082","modelId":"gemini-3.8-flash","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"Partialscore; batchtools;1080p/500steps; Gemini/Sonnet bestof3runs; screenshotonly; officialCUAharness"},{"observationId":"launch-google-gemini-3-8-card-2083","modelId":"gemini-3.7-flash","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"Partialscore; batchtools;1080p/500steps; Gemini/Sonnet bestof3runs; screenshotonly; officialCUAharness"},{"observationId":"launch-google-gemini-3-8-card-2084","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"Partialscore; batchtools;1080p/500steps; Gemini/Sonnet bestof3runs; screenshotonly; officialCUAharness"},{"observationId":"launch-google-gemini-3-8-card-2085","modelId":"claude-sonnet-5","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"Partialscore; batchtools;1080p/500steps; Gemini/Sonnet bestof3runs; screenshotonly; officialCUAharness"},{"observationId":"launch-google-gemini-3-8-card-2086","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"Partialscore; batchtools;1080p/500steps; Gemini/Sonnet bestof3runs; screenshotonly; officialCUAharness"},{"observationId":"launch-google-gemini-3-8-card-2087","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"Partialscore; batchtools;1080p/500steps; Gemini/Sonnet bestof3runs; screenshotonly; officialCUAharness"},{"observationId":"launch-google-gemini-3-8-card-2088","modelId":"gemini-3.8-flash","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"Linuxterminal,bioinfotools,Python,R; allowlistednetwork; Gemini/GPTselfcomputed,Claudeproviderreports"},{"observationId":"launch-google-gemini-3-8-card-2089","modelId":"gemini-3.7-flash","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"Linuxterminal,bioinfotools,Python,R; allowlistednetwork; Gemini/GPTselfcomputed,Claudeproviderreports"},{"observationId":"launch-google-gemini-3-8-card-2090","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"Linuxterminal,bioinfotools,Python,R; allowlistednetwork; Gemini/GPTselfcomputed,Claudeproviderreports"},{"observationId":"launch-google-gemini-3-8-card-2091","modelId":"claude-sonnet-5","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"Linuxterminal,bioinfotools,Python,R; allowlistednetwork; Gemini/GPTselfcomputed,Claudeproviderreports"},{"observationId":"launch-google-gemini-3-8-card-2092","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"Linuxterminal,bioinfotools,Python,R; allowlistednetwork; Gemini/GPTselfcomputed,Claudeproviderreports"},{"observationId":"launch-google-gemini-3-8-card-2093","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"Linuxterminal,bioinfotools,Python,R; allowlistednetwork; Gemini/GPTselfcomputed,Claudeproviderreports"},{"observationId":"launch-google-gemini-3-8-card-2094","modelId":"gemini-3.8-flash","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"Linuxterminal,bioinfotools,Python,R; allowlistednetwork; Gemini/GPTselfcomputed,Claudeproviderreports"},{"observationId":"launch-google-gemini-3-8-card-2095","modelId":"gemini-3.7-flash","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"Linuxterminal,bioinfotools,Python,R; allowlistednetwork; Gemini/GPTselfcomputed,Claudeproviderreports"},{"observationId":"launch-google-gemini-3-8-card-2096","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"Linuxterminal,bioinfotools,Python,R; allowlistednetwork; Gemini/GPTselfcomputed,Claudeproviderreports"},{"observationId":"launch-google-gemini-3-8-card-2097","modelId":"claude-sonnet-5","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"Linuxterminal,bioinfotools,Python,R; allowlistednetwork; Gemini/GPTselfcomputed,Claudeproviderreports"},{"observationId":"launch-google-gemini-3-8-card-2098","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"Linuxterminal,bioinfotools,Python,R; allowlistednetwork; Gemini/GPTselfcomputed,Claudeproviderreports"},{"observationId":"launch-google-gemini-3-8-card-2099","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"Linuxterminal,bioinfotools,Python,R; allowlistednetwork; Gemini/GPTselfcomputed,Claudeproviderreports"},{"observationId":"launch-google-gemini-3-8-card-2100","modelId":"gemini-3.8-flash","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"Selfcomputed; Linuxterminal,bioinfotools,Python,R,network; macroaverage11subtasks"},{"observationId":"launch-google-gemini-3-8-card-2101","modelId":"gemini-3.7-flash","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"Selfcomputed; Linuxterminal,bioinfotools,Python,R,network; macroaverage11subtasks"},{"observationId":"launch-google-gemini-3-8-card-2102","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"Selfcomputed; Linuxterminal,bioinfotools,Python,R,network; macroaverage11subtasks"},{"observationId":"launch-google-gemini-3-8-card-2103","modelId":"claude-sonnet-5","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"Selfcomputed; Linuxterminal,bioinfotools,Python,R,network; macroaverage11subtasks"},{"observationId":"launch-google-gemini-3-8-card-2104","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"Selfcomputed; Linuxterminal,bioinfotools,Python,R,network; macroaverage11subtasks"},{"observationId":"launch-google-gemini-3-8-card-2105","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-3-8-Flash-Model-Card.pdf","configuration":"Selfcomputed; Linuxterminal,bioinfotools,Python,R,network; macroaverage11subtasks"},{"observationId":"launch-kimi-702","modelId":"kimi-k3","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-703","modelId":"claude-fable-5","label":"Max with fallbacks","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-704","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-705","modelId":"claude-opus-4-8","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-706","modelId":"gpt-5.5","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-707","modelId":"glm-5.2","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-708","modelId":"kimi-k3","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-709","modelId":"claude-fable-5","label":"Max with fallbacks","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-710","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-711","modelId":"claude-opus-4-8","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-712","modelId":"gpt-5.5","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-713","modelId":"glm-5.2","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-714","modelId":"kimi-k3","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-715","modelId":"claude-fable-5","label":"Max with fallbacks","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-716","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-717","modelId":"claude-opus-4-8","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-718","modelId":"gpt-5.5","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-719","modelId":"glm-5.2","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-720","modelId":"kimi-k3","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-721","modelId":"kimi-k3","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-722","modelId":"claude-fable-5","label":"Max with fallbacks","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-723","modelId":"claude-fable-5","label":"Max with fallbacks","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-724","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-725","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-726","modelId":"claude-opus-4-8","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-727","modelId":"claude-opus-4-8","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-728","modelId":"gpt-5.5","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-729","modelId":"gpt-5.5","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-730","modelId":"kimi-k3","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-731","modelId":"claude-fable-5","label":"Max with fallbacks","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-732","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-733","modelId":"claude-opus-4-8","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-734","modelId":"gpt-5.5","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-735","modelId":"glm-5.2","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-736","modelId":"kimi-k3","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-737","modelId":"claude-fable-5","label":"Max with fallbacks","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-738","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-739","modelId":"claude-opus-4-8","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-740","modelId":"gpt-5.5","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-741","modelId":"glm-5.2","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-742","modelId":"kimi-k3","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-743","modelId":"claude-fable-5","label":"Max with fallbacks","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-744","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-745","modelId":"claude-opus-4-8","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-746","modelId":"gpt-5.5","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-747","modelId":"glm-5.2","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-748","modelId":"kimi-k3","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-749","modelId":"claude-fable-5","label":"Max with fallbacks","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-750","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-751","modelId":"claude-opus-4-8","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-752","modelId":"gpt-5.5","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-753","modelId":"glm-5.2","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-754","modelId":"kimi-k3","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-755","modelId":"claude-fable-5","label":"Max with fallbacks","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-756","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-757","modelId":"claude-opus-4-8","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-758","modelId":"gpt-5.5","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-759","modelId":"glm-5.2","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-760","modelId":"kimi-k3","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-761","modelId":"claude-fable-5","label":"Max with fallbacks","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-762","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-763","modelId":"claude-opus-4-8","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-764","modelId":"gpt-5.5","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-765","modelId":"glm-5.2","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-766","modelId":"kimi-k3","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-767","modelId":"claude-fable-5","label":"Max with fallbacks","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-768","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-769","modelId":"claude-opus-4-8","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-770","modelId":"gpt-5.5","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-771","modelId":"glm-5.2","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-772","modelId":"kimi-k3","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-773","modelId":"claude-fable-5","label":"Max with fallbacks","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-774","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-775","modelId":"claude-opus-4-8","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-776","modelId":"gpt-5.5","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-777","modelId":"glm-5.2","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-778","modelId":"kimi-k3","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-779","modelId":"claude-fable-5","label":"Max with fallbacks","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-780","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-781","modelId":"claude-opus-4-8","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-782","modelId":"gpt-5.5","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-783","modelId":"glm-5.2","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-784","modelId":"kimi-k3","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-785","modelId":"claude-fable-5","label":"Max with fallbacks","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-786","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-787","modelId":"claude-opus-4-8","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-788","modelId":"gpt-5.5","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-789","modelId":"kimi-k3","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-790","modelId":"claude-fable-5","label":"Max with fallbacks","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-791","modelId":"claude-opus-4-8","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-792","modelId":"kimi-k3","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-793","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-794","modelId":"claude-opus-4-8","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-795","modelId":"gpt-5.5","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-796","modelId":"glm-5.2","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-797","modelId":"kimi-k3","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-798","modelId":"claude-fable-5","label":"Max with fallbacks","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-799","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-800","modelId":"claude-opus-4-8","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-801","modelId":"gpt-5.5","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-802","modelId":"glm-5.2","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-803","modelId":"kimi-k3","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-804","modelId":"claude-fable-5","label":"Max with fallbacks","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-805","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-806","modelId":"claude-opus-4-8","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-807","modelId":"gpt-5.5","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-808","modelId":"glm-5.2","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-809","modelId":"kimi-k3","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-810","modelId":"claude-fable-5","label":"Max with fallbacks","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-811","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-812","modelId":"claude-opus-4-8","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-813","modelId":"gpt-5.5","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-814","modelId":"kimi-k3","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-815","modelId":"claude-fable-5","label":"Max with fallbacks","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-816","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-817","modelId":"claude-opus-4-8","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-818","modelId":"gpt-5.5","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-819","modelId":"glm-5.2","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-820","modelId":"kimi-k3","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-821","modelId":"claude-fable-5","label":"Max with fallbacks","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-822","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-823","modelId":"claude-opus-4-8","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-824","modelId":"gpt-5.5","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-825","modelId":"glm-5.2","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-826","modelId":"kimi-k3","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-827","modelId":"claude-fable-5","label":"Max with fallbacks","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-828","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-829","modelId":"claude-opus-4-8","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-830","modelId":"gpt-5.5","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-831","modelId":"glm-5.2","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-832","modelId":"kimi-k3","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-833","modelId":"claude-fable-5","label":"Max with fallbacks","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-834","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-835","modelId":"claude-opus-4-8","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-836","modelId":"gpt-5.5","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-837","modelId":"glm-5.2","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-838","modelId":"kimi-k3","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-839","modelId":"claude-fable-5","label":"XHigh with fallbacks","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-840","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-841","modelId":"claude-opus-4-8","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-842","modelId":"gpt-5.5","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-843","modelId":"glm-5.2","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-844","modelId":"kimi-k3","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-845","modelId":"claude-fable-5","label":"Max with fallbacks","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-846","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-847","modelId":"claude-opus-4-8","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-848","modelId":"gpt-5.5","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-849","modelId":"glm-5.2","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-850","modelId":"kimi-k3","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-851","modelId":"claude-fable-5","label":"Max with fallbacks","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-852","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-853","modelId":"claude-opus-4-8","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-854","modelId":"gpt-5.5","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-855","modelId":"glm-5.2","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-856","modelId":"kimi-k3","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-857","modelId":"claude-fable-5","label":"Max with fallbacks","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-858","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-859","modelId":"claude-opus-4-8","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-860","modelId":"gpt-5.5","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-861","modelId":"glm-5.2","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-862","modelId":"kimi-k3","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-863","modelId":"claude-fable-5","label":"Max with fallbacks","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-864","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-865","modelId":"claude-opus-4-8","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-866","modelId":"gpt-5.5","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-867","modelId":"kimi-k3","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-868","modelId":"claude-fable-5","label":"Max with fallbacks","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-869","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-870","modelId":"claude-opus-4-8","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-871","modelId":"gpt-5.5","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-872","modelId":"kimi-k3","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-873","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-874","modelId":"claude-opus-4-8","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-875","modelId":"gpt-5.5","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-876","modelId":"kimi-k3","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-877","modelId":"claude-fable-5","label":"Max with fallbacks","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-878","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-879","modelId":"claude-opus-4-8","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-880","modelId":"gpt-5.5","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-881","modelId":"glm-5.2","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-882","modelId":"kimi-k3","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-883","modelId":"claude-fable-5","label":"Max with fallbacks","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-884","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-885","modelId":"claude-opus-4-8","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-886","modelId":"gpt-5.5","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-887","modelId":"glm-5.2","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-888","modelId":"kimi-k3","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-889","modelId":"claude-fable-5","label":"Max with fallbacks","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-890","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-891","modelId":"claude-opus-4-8","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-892","modelId":"gpt-5.5","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-893","modelId":"glm-5.2","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-894","modelId":"kimi-k3","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-895","modelId":"claude-fable-5","label":"Max with fallbacks","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-896","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-897","modelId":"claude-opus-4-8","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-898","modelId":"gpt-5.5","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-899","modelId":"glm-5.2","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-900","modelId":"kimi-k3","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-901","modelId":"claude-fable-5","label":"Max with fallbacks","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-902","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-903","modelId":"claude-opus-4-8","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-904","modelId":"gpt-5.5","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-905","modelId":"glm-5.2","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-906","modelId":"kimi-k3","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-907","modelId":"claude-fable-5","label":"Max with fallbacks","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-908","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-909","modelId":"claude-opus-4-8","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-910","modelId":"gpt-5.5","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-911","modelId":"kimi-k3","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-912","modelId":"claude-fable-5","label":"Max with fallbacks","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-913","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-914","modelId":"claude-opus-4-8","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-915","modelId":"gpt-5.5","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-916","modelId":"kimi-k3","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-917","modelId":"claude-fable-5","label":"Max with fallbacks","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-918","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-919","modelId":"claude-opus-4-8","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-920","modelId":"gpt-5.5","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-921","modelId":"kimi-k3","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-922","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-923","modelId":"claude-opus-4-8","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-924","modelId":"gpt-5.5","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-925","modelId":"kimi-k3","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-926","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-927","modelId":"claude-opus-4-8","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-928","modelId":"gpt-5.5","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-929","modelId":"kimi-k3","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-930","modelId":"claude-fable-5","label":"Max with fallbacks","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-931","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-932","modelId":"claude-opus-4-8","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-933","modelId":"gpt-5.5","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-934","modelId":"kimi-k3","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-935","modelId":"kimi-k3","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-936","modelId":"claude-fable-5","label":"Max with fallbacks","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-937","modelId":"claude-fable-5","label":"Max with fallbacks","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-938","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-939","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-940","modelId":"claude-opus-4-8","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-941","modelId":"claude-opus-4-8","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-942","modelId":"gpt-5.5","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-943","modelId":"gpt-5.5","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-944","modelId":"kimi-k3","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-945","modelId":"kimi-k3","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-946","modelId":"claude-fable-5","label":"Max with fallbacks","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-947","modelId":"claude-fable-5","label":"Max with fallbacks","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-948","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-949","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-950","modelId":"claude-opus-4-8","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-951","modelId":"claude-opus-4-8","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-952","modelId":"gpt-5.5","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-953","modelId":"gpt-5.5","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-954","modelId":"kimi-k3","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-955","modelId":"kimi-k3","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-956","modelId":"claude-fable-5","label":"Max with fallbacks","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-957","modelId":"claude-fable-5","label":"Max with fallbacks","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-958","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-959","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-960","modelId":"claude-opus-4-8","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-961","modelId":"claude-opus-4-8","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-962","modelId":"gpt-5.5","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-963","modelId":"gpt-5.5","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-964","modelId":"kimi-k3","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-965","modelId":"kimi-k3","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-966","modelId":"claude-fable-5","label":"Max with fallbacks","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-967","modelId":"claude-fable-5","label":"Max with fallbacks","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-968","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-969","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-970","modelId":"claude-opus-4-8","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-971","modelId":"claude-opus-4-8","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-972","modelId":"gpt-5.5","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. without tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-973","modelId":"gpt-5.5","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/README.md","configuration":"Table headers: Kimi K3, GPT-5.6 Sol, Opus 4.8 and GLM-5.2 max; GPT-5.5 xhigh; Fable 5 max with fallbacks. Benchmark-specific footnotes override header settings; cited results are not new Moonshot evaluations. Benchmark-specific tools, harness and budget documented under Evaluation Details. with tools Kimi temp1; single-step top_p.95,agentic top_p1; vision tools=Python,3runs except ZeroBench5runs."},{"observationId":"launch-kimi-k26-card-2106","modelId":"kimi-k2.6","label":"Reasoning","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, HLE-Full (w/ tools). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"observationId":"launch-kimi-k26-card-2107","modelId":"gpt-5.4","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, HLE-Full (w/ tools). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"observationId":"launch-kimi-k26-card-2108","modelId":"claude-opus-4-6","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, HLE-Full (w/ tools). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"observationId":"launch-kimi-k26-card-2109","modelId":"gemini-3.1-pro","label":"High","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, HLE-Full (w/ tools). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"observationId":"launch-kimi-k26-card-2110","modelId":"kimi-k2.6","label":"Reasoning","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, BrowseComp. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"observationId":"launch-kimi-k26-card-2111","modelId":"gpt-5.4","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, BrowseComp. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"observationId":"launch-kimi-k26-card-2112","modelId":"claude-opus-4-6","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, BrowseComp. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"observationId":"launch-kimi-k26-card-2113","modelId":"gemini-3.1-pro","label":"High","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, BrowseComp. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"observationId":"launch-kimi-k26-card-2114","modelId":"kimi-k2.6","label":"Reasoning","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, BrowseComp (Agent Swarm). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"observationId":"launch-kimi-k26-card-2115","modelId":"gpt-5.4","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, BrowseComp (Agent Swarm). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"observationId":"launch-kimi-k26-card-2116","modelId":"claude-opus-4-6","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, BrowseComp (Agent Swarm). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"observationId":"launch-kimi-k26-card-2117","modelId":"gemini-3.1-pro","label":"High","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, BrowseComp (Agent Swarm). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"observationId":"launch-kimi-k26-card-2118","modelId":"kimi-k2.6","label":"Reasoning","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, DeepSearchQA (f1-score). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"observationId":"launch-kimi-k26-card-2119","modelId":"gpt-5.4","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, DeepSearchQA (f1-score). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"observationId":"launch-kimi-k26-card-2120","modelId":"claude-opus-4-6","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, DeepSearchQA (f1-score). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"observationId":"launch-kimi-k26-card-2121","modelId":"gemini-3.1-pro","label":"High","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, DeepSearchQA (f1-score). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"observationId":"launch-kimi-k26-card-2122","modelId":"kimi-k2.6","label":"Reasoning","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, DeepSearchQA (accuracy). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"observationId":"launch-kimi-k26-card-2123","modelId":"gpt-5.4","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, DeepSearchQA (accuracy). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"observationId":"launch-kimi-k26-card-2124","modelId":"claude-opus-4-6","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, DeepSearchQA (accuracy). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"observationId":"launch-kimi-k26-card-2125","modelId":"gemini-3.1-pro","label":"High","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, DeepSearchQA (accuracy). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"observationId":"launch-kimi-k26-card-2126","modelId":"kimi-k2.6","label":"Reasoning","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, WideSearch (item-f1). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"observationId":"launch-kimi-k26-card-2127","modelId":"kimi-k2.6","label":"Reasoning","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, Toolathlon. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"observationId":"launch-kimi-k26-card-2128","modelId":"gpt-5.4","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, Toolathlon. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"observationId":"launch-kimi-k26-card-2129","modelId":"claude-opus-4-6","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, Toolathlon. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"observationId":"launch-kimi-k26-card-2130","modelId":"gemini-3.1-pro","label":"High","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, Toolathlon. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"observationId":"launch-kimi-k26-card-2131","modelId":"kimi-k2.6","label":"Reasoning","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, MCPMark. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"observationId":"launch-kimi-k26-card-2132","modelId":"gpt-5.4","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, MCPMark. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"observationId":"launch-kimi-k26-card-2133","modelId":"claude-opus-4-6","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, MCPMark. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"observationId":"launch-kimi-k26-card-2134","modelId":"gemini-3.1-pro","label":"High","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, MCPMark. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"observationId":"launch-kimi-k26-card-2135","modelId":"kimi-k2.6","label":"Reasoning","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, Claw Eval (pass^3). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"observationId":"launch-kimi-k26-card-2136","modelId":"gpt-5.4","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, Claw Eval (pass^3). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"observationId":"launch-kimi-k26-card-2137","modelId":"claude-opus-4-6","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, Claw Eval (pass^3). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"observationId":"launch-kimi-k26-card-2138","modelId":"gemini-3.1-pro","label":"High","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, Claw Eval (pass^3). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"observationId":"launch-kimi-k26-card-2139","modelId":"kimi-k2.6","label":"Reasoning","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, Claw Eval (pass@3). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"observationId":"launch-kimi-k26-card-2140","modelId":"gpt-5.4","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, Claw Eval (pass@3). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"observationId":"launch-kimi-k26-card-2141","modelId":"claude-opus-4-6","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, Claw Eval (pass@3). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"observationId":"launch-kimi-k26-card-2142","modelId":"gemini-3.1-pro","label":"High","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, Claw Eval (pass@3). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"observationId":"launch-kimi-k26-card-2143","modelId":"kimi-k2.6","label":"Reasoning","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, APEX-Agents. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"observationId":"launch-kimi-k26-card-2144","modelId":"gpt-5.4","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, APEX-Agents. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"observationId":"launch-kimi-k26-card-2145","modelId":"claude-opus-4-6","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, APEX-Agents. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"observationId":"launch-kimi-k26-card-2146","modelId":"gemini-3.1-pro","label":"High","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, APEX-Agents. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"observationId":"launch-kimi-k26-card-2147","modelId":"kimi-k2.6","label":"Reasoning","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, OSWorld-Verified. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"observationId":"launch-kimi-k26-card-2148","modelId":"gpt-5.4","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, OSWorld-Verified. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"observationId":"launch-kimi-k26-card-2149","modelId":"claude-opus-4-6","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, OSWorld-Verified. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons."},{"observationId":"launch-kimi-k26-card-2150","modelId":"kimi-k2.6","label":"Reasoning","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, Terminal-Bench 2.0 (Terminus-2). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent."},{"observationId":"launch-kimi-k26-card-2151","modelId":"gpt-5.4","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, Terminal-Bench 2.0 (Terminus-2). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent."},{"observationId":"launch-kimi-k26-card-2152","modelId":"claude-opus-4-6","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, Terminal-Bench 2.0 (Terminus-2). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent."},{"observationId":"launch-kimi-k26-card-2153","modelId":"gemini-3.1-pro","label":"High","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, Terminal-Bench 2.0 (Terminus-2). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent."},{"observationId":"launch-kimi-k26-card-2154","modelId":"kimi-k2.6","label":"Reasoning","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, SWE-Bench Pro. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent."},{"observationId":"launch-kimi-k26-card-2155","modelId":"gpt-5.4","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, SWE-Bench Pro. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent."},{"observationId":"launch-kimi-k26-card-2156","modelId":"claude-opus-4-6","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, SWE-Bench Pro. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent."},{"observationId":"launch-kimi-k26-card-2157","modelId":"gemini-3.1-pro","label":"High","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, SWE-Bench Pro. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent."},{"observationId":"launch-kimi-k26-card-2158","modelId":"kimi-k2.6","label":"Reasoning","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, SWE-Bench Multilingual. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent."},{"observationId":"launch-kimi-k26-card-2159","modelId":"claude-opus-4-6","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, SWE-Bench Multilingual. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent."},{"observationId":"launch-kimi-k26-card-2160","modelId":"gemini-3.1-pro","label":"High","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, SWE-Bench Multilingual. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent."},{"observationId":"launch-kimi-k26-card-2161","modelId":"kimi-k2.6","label":"Reasoning","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, SWE-Bench Verified. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent."},{"observationId":"launch-kimi-k26-card-2162","modelId":"claude-opus-4-6","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, SWE-Bench Verified. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent."},{"observationId":"launch-kimi-k26-card-2163","modelId":"gemini-3.1-pro","label":"High","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, SWE-Bench Verified. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent."},{"observationId":"launch-kimi-k26-card-2164","modelId":"kimi-k2.6","label":"Reasoning","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, SciCode. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent."},{"observationId":"launch-kimi-k26-card-2165","modelId":"gpt-5.4","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, SciCode. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent."},{"observationId":"launch-kimi-k26-card-2166","modelId":"claude-opus-4-6","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, SciCode. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent."},{"observationId":"launch-kimi-k26-card-2167","modelId":"gemini-3.1-pro","label":"High","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, SciCode. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent."},{"observationId":"launch-kimi-k26-card-2168","modelId":"kimi-k2.6","label":"Reasoning","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, OJBench (python). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent."},{"observationId":"launch-kimi-k26-card-2169","modelId":"claude-opus-4-6","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, OJBench (python). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent."},{"observationId":"launch-kimi-k26-card-2170","modelId":"gemini-3.1-pro","label":"High","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, OJBench (python). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent."},{"observationId":"launch-kimi-k26-card-2171","modelId":"kimi-k2.6","label":"Reasoning","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, LiveCodeBench (v6). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent."},{"observationId":"launch-kimi-k26-card-2172","modelId":"claude-opus-4-6","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, LiveCodeBench (v6). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent."},{"observationId":"launch-kimi-k26-card-2173","modelId":"gemini-3.1-pro","label":"High","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, LiveCodeBench (v6). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Coding: average of ten runs; Terminal uses Terminus2 preserve-thinking; SWE uses in-house SWE-agent."},{"observationId":"launch-kimi-k26-card-2174","modelId":"kimi-k2.6","label":"Reasoning","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, HLE-Full. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Reasoning: 98304 generation tokens; HLE full set."},{"observationId":"launch-kimi-k26-card-2175","modelId":"gpt-5.4","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, HLE-Full. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Reasoning: 98304 generation tokens; HLE full set."},{"observationId":"launch-kimi-k26-card-2176","modelId":"claude-opus-4-6","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, HLE-Full. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Reasoning: 98304 generation tokens; HLE full set."},{"observationId":"launch-kimi-k26-card-2177","modelId":"gemini-3.1-pro","label":"High","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, HLE-Full. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Reasoning: 98304 generation tokens; HLE full set."},{"observationId":"launch-kimi-k26-card-2178","modelId":"kimi-k2.6","label":"Reasoning","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, AIME 2026. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Reasoning: 98304 generation tokens; HLE full set."},{"observationId":"launch-kimi-k26-card-2179","modelId":"gpt-5.4","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, AIME 2026. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Reasoning: 98304 generation tokens; HLE full set."},{"observationId":"launch-kimi-k26-card-2180","modelId":"claude-opus-4-6","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, AIME 2026. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Reasoning: 98304 generation tokens; HLE full set."},{"observationId":"launch-kimi-k26-card-2181","modelId":"gemini-3.1-pro","label":"High","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, AIME 2026. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Reasoning: 98304 generation tokens; HLE full set."},{"observationId":"launch-kimi-k26-card-2182","modelId":"kimi-k2.6","label":"Reasoning","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, HMMT 2026 (Feb). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Reasoning: 98304 generation tokens; HLE full set."},{"observationId":"launch-kimi-k26-card-2183","modelId":"gpt-5.4","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, HMMT 2026 (Feb). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Reasoning: 98304 generation tokens; HLE full set."},{"observationId":"launch-kimi-k26-card-2184","modelId":"claude-opus-4-6","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, HMMT 2026 (Feb). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Reasoning: 98304 generation tokens; HLE full set."},{"observationId":"launch-kimi-k26-card-2185","modelId":"gemini-3.1-pro","label":"High","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, HMMT 2026 (Feb). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Reasoning: 98304 generation tokens; HLE full set."},{"observationId":"launch-kimi-k26-card-2186","modelId":"kimi-k2.6","label":"Reasoning","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, IMO-AnswerBench. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Reasoning: 98304 generation tokens; HLE full set."},{"observationId":"launch-kimi-k26-card-2187","modelId":"gpt-5.4","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, IMO-AnswerBench. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Reasoning: 98304 generation tokens; HLE full set."},{"observationId":"launch-kimi-k26-card-2188","modelId":"claude-opus-4-6","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, IMO-AnswerBench. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Reasoning: 98304 generation tokens; HLE full set."},{"observationId":"launch-kimi-k26-card-2189","modelId":"gemini-3.1-pro","label":"High","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, IMO-AnswerBench. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Reasoning: 98304 generation tokens; HLE full set."},{"observationId":"launch-kimi-k26-card-2190","modelId":"kimi-k2.6","label":"Reasoning","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, GPQA-Diamond. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Reasoning: 98304 generation tokens; HLE full set."},{"observationId":"launch-kimi-k26-card-2191","modelId":"gpt-5.4","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, GPQA-Diamond. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Reasoning: 98304 generation tokens; HLE full set."},{"observationId":"launch-kimi-k26-card-2192","modelId":"claude-opus-4-6","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, GPQA-Diamond. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Reasoning: 98304 generation tokens; HLE full set."},{"observationId":"launch-kimi-k26-card-2193","modelId":"gemini-3.1-pro","label":"High","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, GPQA-Diamond. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Reasoning: 98304 generation tokens; HLE full set."},{"observationId":"launch-kimi-k26-card-2194","modelId":"kimi-k2.6","label":"Reasoning","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, MMMU-Pro. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps."},{"observationId":"launch-kimi-k26-card-2195","modelId":"gpt-5.4","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, MMMU-Pro. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps."},{"observationId":"launch-kimi-k26-card-2196","modelId":"claude-opus-4-6","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, MMMU-Pro. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps."},{"observationId":"launch-kimi-k26-card-2197","modelId":"gemini-3.1-pro","label":"High","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, MMMU-Pro. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps."},{"observationId":"launch-kimi-k26-card-2198","modelId":"kimi-k2.6","label":"Reasoning","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, MMMU-Pro (w/ python). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps."},{"observationId":"launch-kimi-k26-card-2199","modelId":"gpt-5.4","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, MMMU-Pro (w/ python). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps."},{"observationId":"launch-kimi-k26-card-2200","modelId":"claude-opus-4-6","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, MMMU-Pro (w/ python). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps."},{"observationId":"launch-kimi-k26-card-2201","modelId":"gemini-3.1-pro","label":"High","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, MMMU-Pro (w/ python). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps."},{"observationId":"launch-kimi-k26-card-2202","modelId":"kimi-k2.6","label":"Reasoning","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, CharXiv (RQ). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps."},{"observationId":"launch-kimi-k26-card-2203","modelId":"gpt-5.4","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, CharXiv (RQ). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps."},{"observationId":"launch-kimi-k26-card-2204","modelId":"claude-opus-4-6","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, CharXiv (RQ). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps."},{"observationId":"launch-kimi-k26-card-2205","modelId":"gemini-3.1-pro","label":"High","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, CharXiv (RQ). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps."},{"observationId":"launch-kimi-k26-card-2206","modelId":"kimi-k2.6","label":"Reasoning","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, CharXiv (RQ) (w/ python). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps."},{"observationId":"launch-kimi-k26-card-2207","modelId":"gpt-5.4","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, CharXiv (RQ) (w/ python). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps."},{"observationId":"launch-kimi-k26-card-2208","modelId":"claude-opus-4-6","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, CharXiv (RQ) (w/ python). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps."},{"observationId":"launch-kimi-k26-card-2209","modelId":"gemini-3.1-pro","label":"High","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, CharXiv (RQ) (w/ python). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps."},{"observationId":"launch-kimi-k26-card-2210","modelId":"kimi-k2.6","label":"Reasoning","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, MathVision. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps."},{"observationId":"launch-kimi-k26-card-2211","modelId":"gpt-5.4","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, MathVision. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps."},{"observationId":"launch-kimi-k26-card-2212","modelId":"claude-opus-4-6","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, MathVision. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps."},{"observationId":"launch-kimi-k26-card-2213","modelId":"gemini-3.1-pro","label":"High","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, MathVision. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps."},{"observationId":"launch-kimi-k26-card-2214","modelId":"kimi-k2.6","label":"Reasoning","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, MathVision (w/ python). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps."},{"observationId":"launch-kimi-k26-card-2215","modelId":"gpt-5.4","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, MathVision (w/ python). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps."},{"observationId":"launch-kimi-k26-card-2216","modelId":"claude-opus-4-6","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, MathVision (w/ python). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps."},{"observationId":"launch-kimi-k26-card-2217","modelId":"gemini-3.1-pro","label":"High","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, MathVision (w/ python). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps."},{"observationId":"launch-kimi-k26-card-2218","modelId":"kimi-k2.6","label":"Reasoning","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, BabyVision. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps."},{"observationId":"launch-kimi-k26-card-2219","modelId":"gpt-5.4","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, BabyVision. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps."},{"observationId":"launch-kimi-k26-card-2220","modelId":"claude-opus-4-6","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, BabyVision. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps."},{"observationId":"launch-kimi-k26-card-2221","modelId":"gemini-3.1-pro","label":"High","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, BabyVision. Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps."},{"observationId":"launch-kimi-k26-card-2222","modelId":"kimi-k2.6","label":"Reasoning","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, BabyVision (w/ python). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps."},{"observationId":"launch-kimi-k26-card-2223","modelId":"gpt-5.4","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, BabyVision (w/ python). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps."},{"observationId":"launch-kimi-k26-card-2224","modelId":"claude-opus-4-6","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, BabyVision (w/ python). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps."},{"observationId":"launch-kimi-k26-card-2225","modelId":"gemini-3.1-pro","label":"High","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, BabyVision (w/ python). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps."},{"observationId":"launch-kimi-k26-card-2226","modelId":"kimi-k2.6","label":"Reasoning","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, V* (w/ python). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps."},{"observationId":"launch-kimi-k26-card-2227","modelId":"gpt-5.4","label":"XHigh","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, V* (w/ python). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps."},{"observationId":"launch-kimi-k26-card-2228","modelId":"claude-opus-4-6","label":"Max","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, V* (w/ python). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps."},{"observationId":"launch-kimi-k26-card-2229","modelId":"gemini-3.1-pro","label":"High","sourceUrl":"https://huggingface.co/moonshotai/Kimi-K2.6/blob/7eb5002f6aadc958aed6a9177b7ed26bb94011bb/README.md","configuration":"Kimi K2.6 card, V* (w/ python). Thinking mode; temperature 1.0, top-p 1.0, context 262144. Comparator settings: GPT-5.4 xhigh, Opus4.6 max, Gemini3.1Pro high. See benchmark-specific footnotes. Only own K2.6 and starred re-evaluations can form new comparisons. Vision: 98304 tokens, average of three runs; Python condition uses 65536 tokens per step, at most50steps."},{"observationId":"launch-meta-1875","modelId":"muse-spark-1.1","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1876","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1877","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1878","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1879","modelId":"muse-spark-1.1","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1880","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1881","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1882","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1883","modelId":"muse-spark-1.1","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1884","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1885","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1886","modelId":"muse-spark-1.1","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1887","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1888","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1889","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1890","modelId":"muse-spark-1.1","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1891","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1892","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1893","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1894","modelId":"muse-spark-1.1","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1895","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1896","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1897","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1898","modelId":"muse-spark-1.1","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1899","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1900","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1901","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1902","modelId":"muse-spark-1.1","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1903","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1904","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1905","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1906","modelId":"muse-spark-1.1","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1907","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1908","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1909","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1910","modelId":"muse-spark-1.1","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1911","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1912","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1913","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1914","modelId":"muse-spark-1.1","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1915","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1916","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1917","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1918","modelId":"muse-spark-1.1","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1919","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1920","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1921","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1922","modelId":"muse-spark-1.1","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1923","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1924","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1925","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1926","modelId":"muse-spark-1.1","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1927","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1928","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1929","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1930","modelId":"muse-spark-1.1","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1931","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1932","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1933","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1934","modelId":"muse-spark-1.1","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1935","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1936","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1937","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1938","modelId":"muse-spark-1.1","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1939","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1940","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1941","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1942","modelId":"muse-spark-1.1","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1943","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1944","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1945","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1946","modelId":"muse-spark-1.1","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1947","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1948","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-1949","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-1-evaluation-report","configuration":"Meta Model API xhigh; Gemini high; Opus max; GPT xhigh. Best of self-reported or Meta reproduction unless specified. OSWorld GUI only; Terminal bash only5attempts89tasks6CPU8GB; DeepSWE no internet5attempts."},{"observationId":"launch-meta-muse-1-3-report-1984","modelId":"muse-spark-1.3","label":"Max","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"Artificial Analysis Stirrup shell/web harness;220tasks; blind pairwise comparisons; reasoning max"},{"observationId":"launch-meta-muse-1-3-report-1985","modelId":"muse-spark-1.3","label":"XHigh","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"Artificial Analysis Stirrup shell/web harness;220tasks; blind pairwise comparisons; reasoning xhigh"},{"observationId":"launch-meta-muse-1-3-report-1986","modelId":"muse-spark-1.2","label":"XHigh","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"Artificial Analysis Stirrup shell/web harness;220tasks; blind pairwise comparisons; reasoning xhigh"},{"observationId":"launch-meta-muse-1-3-report-1987","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"Artificial Analysis Stirrup shell/web harness;220tasks; blind pairwise comparisons; reasoning max"},{"observationId":"launch-meta-muse-1-3-report-1988","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"Artificial Analysis Stirrup shell/web harness;220tasks; blind pairwise comparisons; reasoning max"},{"observationId":"launch-meta-muse-1-3-report-1989","modelId":"muse-spark-1.3","label":"Max","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"65tasks; mean rubric score; official OpenCode harness and file-aware grader; reasoning max"},{"observationId":"launch-meta-muse-1-3-report-1990","modelId":"muse-spark-1.3","label":"XHigh","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"65tasks; mean rubric score; official OpenCode harness and file-aware grader; reasoning xhigh"},{"observationId":"launch-meta-muse-1-3-report-1991","modelId":"muse-spark-1.2","label":"XHigh","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"65tasks; mean rubric score; official OpenCode harness and file-aware grader; reasoning xhigh"},{"observationId":"launch-meta-muse-1-3-report-1992","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"65tasks; mean rubric score; official OpenCode harness and file-aware grader; reasoning max"},{"observationId":"launch-meta-muse-1-3-report-1993","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"65tasks; mean rubric score; official OpenCode harness and file-aware grader; reasoning max"},{"observationId":"launch-meta-muse-1-3-report-1994","modelId":"muse-spark-1.3","label":"Max","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"108tasks; common internal GUI framework; execution-based checkers; partial metric; reasoning max"},{"observationId":"launch-meta-muse-1-3-report-1995","modelId":"muse-spark-1.3","label":"XHigh","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"108tasks; common internal GUI framework; execution-based checkers; partial metric; reasoning xhigh"},{"observationId":"launch-meta-muse-1-3-report-1996","modelId":"muse-spark-1.2","label":"XHigh","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"108tasks; common internal GUI framework; execution-based checkers; partial metric; reasoning xhigh"},{"observationId":"launch-meta-muse-1-3-report-1997","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"108tasks; common internal GUI framework; execution-based checkers; partial metric; reasoning max"},{"observationId":"launch-meta-muse-1-3-report-1998","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"108tasks; common internal GUI framework; execution-based checkers; partial metric; reasoning max"},{"observationId":"launch-meta-muse-1-3-report-1999","modelId":"muse-spark-1.3","label":"Max","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"108tasks; common internal GUI framework; execution-based checkers; binary metric; reasoning max"},{"observationId":"launch-meta-muse-1-3-report-2000","modelId":"muse-spark-1.3","label":"XHigh","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"108tasks; common internal GUI framework; execution-based checkers; binary metric; reasoning xhigh"},{"observationId":"launch-meta-muse-1-3-report-2001","modelId":"muse-spark-1.2","label":"XHigh","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"108tasks; common internal GUI framework; execution-based checkers; binary metric; reasoning xhigh"},{"observationId":"launch-meta-muse-1-3-report-2002","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"108tasks; common internal GUI framework; execution-based checkers; binary metric; reasoning max"},{"observationId":"launch-meta-muse-1-3-report-2003","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"108tasks; common internal GUI framework; execution-based checkers; binary metric; reasoning max"},{"observationId":"launch-meta-muse-1-3-report-2004","modelId":"muse-spark-1.3","label":"Max","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"900questions; common search backend/browser harness; answer-set F1; reasoning max"},{"observationId":"launch-meta-muse-1-3-report-2005","modelId":"muse-spark-1.3","label":"XHigh","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"900questions; common search backend/browser harness; answer-set F1; reasoning xhigh"},{"observationId":"launch-meta-muse-1-3-report-2006","modelId":"muse-spark-1.2","label":"XHigh","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"900questions; common search backend/browser harness; answer-set F1; reasoning xhigh"},{"observationId":"launch-meta-muse-1-3-report-2007","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"900questions; common search backend/browser harness; answer-set F1; reasoning max"},{"observationId":"launch-meta-muse-1-3-report-2008","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"900questions; common search backend/browser harness; answer-set F1; reasoning max"},{"observationId":"launch-meta-muse-1-3-report-2009","modelId":"muse-spark-1.3","label":"Max","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"Internal composite instruction-following evaluations; no fixed task count; reasoning max"},{"observationId":"launch-meta-muse-1-3-report-2010","modelId":"muse-spark-1.3","label":"XHigh","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"Internal composite instruction-following evaluations; no fixed task count; reasoning xhigh"},{"observationId":"launch-meta-muse-1-3-report-2011","modelId":"muse-spark-1.2","label":"XHigh","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"Internal composite instruction-following evaluations; no fixed task count; reasoning xhigh"},{"observationId":"launch-meta-muse-1-3-report-2012","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"Internal composite instruction-following evaluations; no fixed task count; reasoning max"},{"observationId":"launch-meta-muse-1-3-report-2013","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"Internal composite instruction-following evaluations; no fixed task count; reasoning max"},{"observationId":"launch-meta-muse-1-3-report-2014","modelId":"muse-spark-1.3","label":"Max","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"600public workflow tasks; deterministic end-state assertions;pass@1; reasoning max"},{"observationId":"launch-meta-muse-1-3-report-2015","modelId":"muse-spark-1.3","label":"XHigh","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"600public workflow tasks; deterministic end-state assertions;pass@1; reasoning xhigh"},{"observationId":"launch-meta-muse-1-3-report-2016","modelId":"muse-spark-1.2","label":"XHigh","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"600public workflow tasks; deterministic end-state assertions;pass@1; reasoning xhigh"},{"observationId":"launch-meta-muse-1-3-report-2017","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"600public workflow tasks; deterministic end-state assertions;pass@1; reasoning max"},{"observationId":"launch-meta-muse-1-3-report-2018","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"600public workflow tasks; deterministic end-state assertions;pass@1; reasoning max"},{"observationId":"launch-meta-muse-1-3-report-2019","modelId":"muse-spark-1.3","label":"Max","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"8needle;100examples/band rebinned byo200k_base; no tools; sequence-matcher ratio; GPT fromOpenAIcard; reasoning max"},{"observationId":"launch-meta-muse-1-3-report-2020","modelId":"muse-spark-1.3","label":"XHigh","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"8needle;100examples/band rebinned byo200k_base; no tools; sequence-matcher ratio; GPT fromOpenAIcard; reasoning xhigh"},{"observationId":"launch-meta-muse-1-3-report-2021","modelId":"muse-spark-1.2","label":"XHigh","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"8needle;100examples/band rebinned byo200k_base; no tools; sequence-matcher ratio; GPT fromOpenAIcard; reasoning xhigh"},{"observationId":"launch-meta-muse-1-3-report-2022","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"8needle;100examples/band rebinned byo200k_base; no tools; sequence-matcher ratio; GPT fromOpenAIcard; reasoning max"},{"observationId":"launch-meta-muse-1-3-report-2023","modelId":"muse-spark-1.3","label":"Max","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"8needle;100examples/band rebinned byo200k_base; no tools; sequence-matcher ratio; GPT fromOpenAIcard; reasoning max"},{"observationId":"launch-meta-muse-1-3-report-2024","modelId":"muse-spark-1.3","label":"XHigh","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"8needle;100examples/band rebinned byo200k_base; no tools; sequence-matcher ratio; GPT fromOpenAIcard; reasoning xhigh"},{"observationId":"launch-meta-muse-1-3-report-2025","modelId":"muse-spark-1.2","label":"XHigh","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"8needle;100examples/band rebinned byo200k_base; no tools; sequence-matcher ratio; GPT fromOpenAIcard; reasoning xhigh"},{"observationId":"launch-meta-muse-1-3-report-2026","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"8needle;100examples/band rebinned byo200k_base; no tools; sequence-matcher ratio; GPT fromOpenAIcard; reasoning max"},{"observationId":"launch-meta-muse-1-3-report-2027","modelId":"muse-spark-1.3","label":"Max","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"113tasks; Muse1.3mini-swe-agent; comparators officialDatacurve board; reasoning max"},{"observationId":"launch-meta-muse-1-3-report-2028","modelId":"muse-spark-1.2","label":"XHigh","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"113tasks; Muse1.3mini-swe-agent; comparators officialDatacurve board; reasoning xhigh"},{"observationId":"launch-meta-muse-1-3-report-2029","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"113tasks; Muse1.3mini-swe-agent; comparators officialDatacurve board; reasoning max"},{"observationId":"launch-meta-muse-1-3-report-2030","modelId":"muse-spark-1.3","label":"Max","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"124tasks/11repos; publicQnA mini-swe-agent rubric pass@1; GPT/Opus owncards; reasoning max"},{"observationId":"launch-meta-muse-1-3-report-2031","modelId":"muse-spark-1.3","label":"XHigh","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"124tasks/11repos; publicQnA mini-swe-agent rubric pass@1; GPT/Opus owncards; reasoning xhigh"},{"observationId":"launch-meta-muse-1-3-report-2032","modelId":"muse-spark-1.2","label":"XHigh","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"124tasks/11repos; publicQnA mini-swe-agent rubric pass@1; GPT/Opus owncards; reasoning xhigh"},{"observationId":"launch-meta-muse-1-3-report-2033","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"124tasks/11repos; publicQnA mini-swe-agent rubric pass@1; GPT/Opus owncards; reasoning max"},{"observationId":"launch-meta-muse-1-3-report-2034","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"124tasks/11repos; publicQnA mini-swe-agent rubric pass@1; GPT/Opus owncards; reasoning max"},{"observationId":"launch-meta-muse-1-3-report-2035","modelId":"muse-spark-1.3","label":"Max","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"89tasks; each model nativecoding harness in internalframework/cloudsandbox; officialverifier; GPT owncard; reasoning max; native harness family: Meta (exact harness revision not specified)"},{"observationId":"launch-meta-muse-1-3-report-2036","modelId":"muse-spark-1.3","label":"XHigh","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"89tasks; each model nativecoding harness in internalframework/cloudsandbox; officialverifier; GPT owncard; reasoning xhigh; native harness family: Meta (exact harness revision not specified)"},{"observationId":"launch-meta-muse-1-3-report-2037","modelId":"muse-spark-1.2","label":"XHigh","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"89tasks; each model nativecoding harness in internalframework/cloudsandbox; officialverifier; GPT owncard; reasoning xhigh; native harness family: Meta (exact harness revision not specified)"},{"observationId":"launch-meta-muse-1-3-report-2038","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"89tasks; each model nativecoding harness in internalframework/cloudsandbox; officialverifier; GPT owncard; reasoning max; native harness family: OpenAI (exact harness revision not specified)"},{"observationId":"launch-meta-muse-1-3-report-2039","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://research.meta.ai/static/muse-spark-1-3-multimodal-evaluation-methodology","configuration":"89tasks; each model nativecoding harness in internalframework/cloudsandbox; officialverifier; GPT owncard; reasoning max; native harness family: Anthropic (exact harness revision not specified)"},{"observationId":"launch-microsoft-1375","modelId":"mai-thinking-1","label":"Not specified","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports."},{"observationId":"launch-microsoft-1376","modelId":"claude-sonnet-4-6","label":"Not specified","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports."},{"observationId":"launch-microsoft-1377","modelId":"claude-opus-4-6","label":"Not specified","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports."},{"observationId":"launch-microsoft-1378","modelId":"mai-thinking-1","label":"Not specified","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports."},{"observationId":"launch-microsoft-1379","modelId":"kimi-k2.6","label":"Not specified","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports."},{"observationId":"launch-microsoft-1380","modelId":"glm-5.1","label":"Not specified","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports."},{"observationId":"launch-microsoft-1381","modelId":"mai-thinking-1","label":"Not specified","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports."},{"observationId":"launch-microsoft-1382","modelId":"kimi-k2.6","label":"Not specified","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports."},{"observationId":"launch-microsoft-1383","modelId":"glm-5.1","label":"Not specified","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports."},{"observationId":"launch-microsoft-1384","modelId":"mai-thinking-1","label":"Not specified","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports."},{"observationId":"launch-microsoft-1385","modelId":"claude-sonnet-4-6","label":"Not specified","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports."},{"observationId":"launch-microsoft-1386","modelId":"claude-opus-4-6","label":"Not specified","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports."},{"observationId":"launch-microsoft-1387","modelId":"gpt-5.4","label":"Not specified","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports."},{"observationId":"launch-microsoft-1388","modelId":"kimi-k2.6","label":"Not specified","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports."},{"observationId":"launch-microsoft-1389","modelId":"glm-5.1","label":"Not specified","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports."},{"observationId":"launch-microsoft-1390","modelId":"mai-thinking-1","label":"Not specified","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports."},{"observationId":"launch-microsoft-1391","modelId":"kimi-k2.6","label":"Not specified","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports."},{"observationId":"launch-microsoft-1392","modelId":"mai-thinking-1","label":"Not specified","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports."},{"observationId":"launch-microsoft-1393","modelId":"claude-sonnet-4-6","label":"Not specified","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports."},{"observationId":"launch-microsoft-1394","modelId":"claude-opus-4-6","label":"Not specified","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports."},{"observationId":"launch-microsoft-1395","modelId":"gpt-5.4","label":"Not specified","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports."},{"observationId":"launch-microsoft-1396","modelId":"kimi-k2.6","label":"Not specified","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports."},{"observationId":"launch-microsoft-1397","modelId":"glm-5.1","label":"Not specified","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports."},{"observationId":"launch-microsoft-1398","modelId":"mai-thinking-1","label":"Not specified","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports."},{"observationId":"launch-microsoft-1399","modelId":"claude-sonnet-4-6","label":"Not specified","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports."},{"observationId":"launch-microsoft-1400","modelId":"claude-opus-4-6","label":"Not specified","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports."},{"observationId":"launch-microsoft-1401","modelId":"kimi-k2.6","label":"Not specified","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports."},{"observationId":"launch-microsoft-1402","modelId":"mai-thinking-1","label":"Not specified","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports."},{"observationId":"launch-microsoft-1403","modelId":"claude-opus-4-6","label":"Not specified","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports."},{"observationId":"launch-microsoft-1404","modelId":"gpt-5.4","label":"Not specified","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports."},{"observationId":"launch-microsoft-1405","modelId":"kimi-k2.6","label":"Not specified","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports."},{"observationId":"launch-microsoft-1406","modelId":"glm-5.1","label":"Not specified","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf","configuration":"MAI avg 4 runs, temperature 1, top-p .97; simple ReAct bash/string-replace; Terminal-Bench ignores predefined timeouts. Comparator configurations from cited reports."},{"observationId":"launch-microsoft-1407","modelId":"mai-thinking-1","label":"Not specified","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf","configuration":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs."},{"observationId":"launch-microsoft-1408","modelId":"claude-sonnet-4-6","label":"Maximum reasoning","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf","configuration":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs."},{"observationId":"launch-microsoft-1409","modelId":"mai-thinking-1","label":"Not specified","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf","configuration":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs."},{"observationId":"launch-microsoft-1410","modelId":"claude-sonnet-4-6","label":"Maximum reasoning","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf","configuration":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs."},{"observationId":"launch-microsoft-1411","modelId":"mai-thinking-1","label":"Not specified","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf","configuration":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs."},{"observationId":"launch-microsoft-1412","modelId":"claude-sonnet-4-6","label":"Maximum reasoning","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf","configuration":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs."},{"observationId":"launch-microsoft-1413","modelId":"mai-thinking-1","label":"Not specified","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf","configuration":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs."},{"observationId":"launch-microsoft-1414","modelId":"claude-sonnet-4-6","label":"Maximum reasoning","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf","configuration":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs."},{"observationId":"launch-microsoft-1415","modelId":"mai-thinking-1","label":"Not specified","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf","configuration":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs."},{"observationId":"launch-microsoft-1416","modelId":"claude-sonnet-4-6","label":"Maximum reasoning","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf","configuration":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs."},{"observationId":"launch-microsoft-1417","modelId":"mai-thinking-1","label":"Not specified","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf","configuration":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs."},{"observationId":"launch-microsoft-1418","modelId":"claude-sonnet-4-6","label":"Maximum reasoning","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf","configuration":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs."},{"observationId":"launch-microsoft-1419","modelId":"mai-thinking-1","label":"Not specified","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf","configuration":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs."},{"observationId":"launch-microsoft-1420","modelId":"claude-sonnet-4-6","label":"Maximum reasoning","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf","configuration":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs."},{"observationId":"launch-microsoft-1421","modelId":"mai-thinking-1","label":"Not specified","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf","configuration":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs."},{"observationId":"launch-microsoft-1422","modelId":"claude-sonnet-4-6","label":"Maximum reasoning","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf","configuration":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs."},{"observationId":"launch-microsoft-1423","modelId":"mai-thinking-1","label":"Not specified","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf","configuration":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs."},{"observationId":"launch-microsoft-1424","modelId":"claude-sonnet-4-6","label":"Maximum reasoning","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf","configuration":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs."},{"observationId":"launch-microsoft-1425","modelId":"mai-thinking-1","label":"Not specified","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf","configuration":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs."},{"observationId":"launch-microsoft-1426","modelId":"claude-sonnet-4-6","label":"Maximum reasoning","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf","configuration":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs."},{"observationId":"launch-microsoft-1427","modelId":"mai-thinking-1","label":"Not specified","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf","configuration":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs."},{"observationId":"launch-microsoft-1428","modelId":"claude-sonnet-4-6","label":"Maximum reasoning","sourceUrl":"https://microsoft.ai/pdf/mai-thinking-1.pdf","configuration":"Microsoft own evaluation suite; Sonnet max reasoning; Tables 12/19. LongBenchV2 capped 256k,408 questions; CorpusQA GPT-5.4 high judge; 4 runs."},{"observationId":"launch-minimax-1692","modelId":"minimax-m3","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1693","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1694","modelId":"claude-opus-4-7","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1695","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1696","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1697","modelId":"claude-sonnet-4-6","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1698","modelId":"kimi-k2.6","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1699","modelId":"minimax-m3","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1700","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1701","modelId":"claude-opus-4-7","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1702","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1703","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1704","modelId":"glm-5.1","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1705","modelId":"kimi-k2.6","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1706","modelId":"minimax-m3","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1707","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1708","modelId":"claude-opus-4-7","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1709","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1710","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1711","modelId":"glm-5.1","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1712","modelId":"kimi-k2.6","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1713","modelId":"minimax-m3","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1714","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1715","modelId":"claude-opus-4-7","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1716","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1717","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1718","modelId":"claude-sonnet-4-6","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1719","modelId":"minimax-m3","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1720","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1721","modelId":"claude-opus-4-7","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1722","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1723","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1724","modelId":"glm-5.1","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1725","modelId":"kimi-k2.6","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1726","modelId":"minimax-m3","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1727","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1728","modelId":"claude-opus-4-7","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1729","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1730","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1731","modelId":"claude-sonnet-4-6","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1732","modelId":"minimax-m3","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1733","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1734","modelId":"claude-opus-4-7","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1735","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1736","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1737","modelId":"minimax-m3","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1738","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1739","modelId":"claude-opus-4-7","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1740","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1741","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1742","modelId":"minimax-m3","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1743","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1744","modelId":"claude-opus-4-7","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1745","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1746","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1747","modelId":"minimax-m3","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1748","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1749","modelId":"claude-opus-4-7","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1750","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1751","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1752","modelId":"claude-sonnet-4-6","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1753","modelId":"glm-5.1","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1754","modelId":"kimi-k2.6","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1755","modelId":"minimax-m3","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1756","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1757","modelId":"claude-opus-4-7","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1758","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1759","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1760","modelId":"claude-sonnet-4-6","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1761","modelId":"glm-5.1","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1762","modelId":"kimi-k2.6","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1763","modelId":"minimax-m3","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1764","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1765","modelId":"claude-opus-4-7","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1766","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1767","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1768","modelId":"minimax-m3","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1769","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1770","modelId":"claude-opus-4-7","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1771","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1772","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1773","modelId":"minimax-m3","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1774","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1775","modelId":"claude-opus-4-7","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1776","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1777","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1778","modelId":"minimax-m3","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1779","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1780","modelId":"claude-opus-4-7","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1781","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1782","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1783","modelId":"claude-sonnet-4-6","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1784","modelId":"glm-5.1","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1785","modelId":"kimi-k2.6","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1786","modelId":"minimax-m3","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1787","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1788","modelId":"claude-opus-4-7","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1789","modelId":"claude-sonnet-4-6","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1790","modelId":"minimax-m3","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1791","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1792","modelId":"claude-opus-4-7","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1793","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1794","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1795","modelId":"claude-sonnet-4-6","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1796","modelId":"glm-5.1","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1797","modelId":"kimi-k2.6","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1798","modelId":"minimax-m3","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1799","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1800","modelId":"claude-opus-4-7","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1801","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1802","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1803","modelId":"minimax-m3","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1804","modelId":"claude-opus-4-7","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1805","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1806","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1807","modelId":"minimax-m3","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1808","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1809","modelId":"claude-opus-4-7","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1810","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1811","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1812","modelId":"glm-5.1","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1813","modelId":"kimi-k2.6","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1814","modelId":"minimax-m3","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1815","modelId":"claude-opus-4-7","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1816","modelId":"minimax-m3","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1817","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1818","modelId":"claude-opus-4-7","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1819","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1820","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1821","modelId":"claude-sonnet-4-6","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1822","modelId":"glm-5.1","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1823","modelId":"kimi-k2.6","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1824","modelId":"minimax-m3","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1825","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1826","modelId":"claude-opus-4-7","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1827","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1828","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1829","modelId":"claude-sonnet-4-6","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1830","modelId":"minimax-m3","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1831","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1832","modelId":"claude-opus-4-7","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1833","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1834","modelId":"claude-sonnet-4-6","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1835","modelId":"glm-5.1","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1836","modelId":"kimi-k2.6","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1837","modelId":"minimax-m3","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1838","modelId":"claude-opus-4-7","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1839","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1840","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1841","modelId":"claude-sonnet-4-6","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1842","modelId":"kimi-k2.6","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1843","modelId":"minimax-m3","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1844","modelId":"claude-opus-4-7","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1845","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1846","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1847","modelId":"claude-sonnet-4-6","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1848","modelId":"minimax-m3","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1849","modelId":"claude-opus-4-7","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1850","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1851","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1852","modelId":"claude-sonnet-4-6","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1853","modelId":"kimi-k2.6","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1854","modelId":"minimax-m3","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1855","modelId":"claude-opus-4-7","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1856","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1857","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1858","modelId":"minimax-m3","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1859","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1860","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax launch figure methodology. SWE internal ClaudeCode 4 runs; Terminal 8CPU16GB2h128k Terminus2; Paper/PostTrain Ralph-loop12h; GDPval internal rubric grader; BrowseComp discard64k; OSWorld361tasks. Comparator provider/leaderboard scores where footnoted."},{"observationId":"launch-minimax-1861","modelId":"minimax-m3","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"Launch figure; agent final assets"},{"observationId":"launch-minimax-1862","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"Launch figure; agent final assets"},{"observationId":"launch-minimax-1863","modelId":"claude-opus-4-7","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"Launch figure; agent final assets"},{"observationId":"launch-minimax-1864","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"Launch figure; agent final assets"},{"observationId":"launch-minimax-1865","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"Launch figure; agent final assets"},{"observationId":"launch-minimax-1866","modelId":"claude-sonnet-4-6","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"Launch figure; agent final assets"},{"observationId":"launch-minimax-1867","modelId":"minimax-m3","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax points out of42; comparator percentages as printed"},{"observationId":"launch-minimax-1868","modelId":"claude-opus-4-7","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax points out of42; comparator percentages as printed"},{"observationId":"launch-minimax-1869","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax points out of42; comparator percentages as printed"},{"observationId":"launch-minimax-1870","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax points out of42; comparator percentages as printed"},{"observationId":"launch-minimax-1871","modelId":"minimax-m3","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax points out of42; comparator percentages as printed"},{"observationId":"launch-minimax-1872","modelId":"claude-opus-4-7","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax points out of42; comparator percentages as printed"},{"observationId":"launch-minimax-1873","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax points out of42; comparator percentages as printed"},{"observationId":"launch-minimax-1874","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configuration":"MiniMax points out of42; comparator percentages as printed"},{"observationId":"launch-mistral-1442","modelId":"mistral-medium-3.5","label":"Maximum reasoning","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; Sonnet4.6 external API truncation caveat."},{"observationId":"launch-mistral-1443","modelId":"claude-sonnet-4-5-20250929","label":"Not specified","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; Sonnet4.6 external API truncation caveat."},{"observationId":"launch-mistral-1444","modelId":"claude-sonnet-4-6","label":"Not specified","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; Sonnet4.6 external API truncation caveat."},{"observationId":"launch-mistral-1445","modelId":"qwen3.5-397b-a17b","label":"Not specified","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; Sonnet4.6 external API truncation caveat."},{"observationId":"launch-mistral-1446","modelId":"glm-5","label":"Not specified","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; Sonnet4.6 external API truncation caveat."},{"observationId":"launch-mistral-1447","modelId":"mistral-medium-3.5","label":"Maximum reasoning","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; Sonnet4.6 external API truncation caveat."},{"observationId":"launch-mistral-1448","modelId":"claude-sonnet-4-5-20250929","label":"Not specified","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; Sonnet4.6 external API truncation caveat."},{"observationId":"launch-mistral-1449","modelId":"claude-sonnet-4-6","label":"Not specified","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; Sonnet4.6 external API truncation caveat."},{"observationId":"launch-mistral-1450","modelId":"qwen3.5-397b-a17b","label":"Not specified","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; Sonnet4.6 external API truncation caveat."},{"observationId":"launch-mistral-1451","modelId":"glm-5","label":"Not specified","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; Sonnet4.6 external API truncation caveat."},{"observationId":"launch-mistral-1452","modelId":"mistral-medium-3.5","label":"Maximum reasoning","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; Sonnet4.6 external API truncation caveat."},{"observationId":"launch-mistral-1453","modelId":"claude-sonnet-4-5-20250929","label":"Not specified","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; Sonnet4.6 external API truncation caveat."},{"observationId":"launch-mistral-1454","modelId":"claude-sonnet-4-6","label":"Not specified","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; Sonnet4.6 external API truncation caveat."},{"observationId":"launch-mistral-1455","modelId":"qwen3.5-397b-a17b","label":"Not specified","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; Sonnet4.6 external API truncation caveat."},{"observationId":"launch-mistral-1456","modelId":"glm-5","label":"Not specified","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; Sonnet4.6 external API truncation caveat."},{"observationId":"launch-mistral-1457","modelId":"mistral-medium-3.5","label":"Maximum reasoning","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; Sonnet4.6 external API truncation caveat."},{"observationId":"launch-mistral-1458","modelId":"claude-sonnet-4-5-20250929","label":"Not specified","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; Sonnet4.6 external API truncation caveat."},{"observationId":"launch-mistral-1459","modelId":"claude-sonnet-4-6","label":"Not specified","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; Sonnet4.6 external API truncation caveat."},{"observationId":"launch-mistral-1460","modelId":"qwen3.5-397b-a17b","label":"Not specified","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; Sonnet4.6 external API truncation caveat."},{"observationId":"launch-mistral-1461","modelId":"mistral-medium-3.5","label":"Maximum reasoning","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k."},{"observationId":"launch-mistral-1462","modelId":"claude-sonnet-4-5-20250929","label":"Not specified","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k."},{"observationId":"launch-mistral-1463","modelId":"claude-sonnet-4-6","label":"Not specified","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k."},{"observationId":"launch-mistral-1464","modelId":"glm-5.1","label":"Not specified","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k."},{"observationId":"launch-mistral-1465","modelId":"qwen3.5-397b-a17b","label":"Not specified","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k."},{"observationId":"launch-mistral-1466","modelId":"mistral-medium-3.5","label":"Maximum reasoning","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k."},{"observationId":"launch-mistral-1467","modelId":"claude-sonnet-4-5-20250929","label":"Not specified","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k."},{"observationId":"launch-mistral-1468","modelId":"claude-sonnet-4-6","label":"Not specified","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k."},{"observationId":"launch-mistral-1469","modelId":"glm-5.1","label":"Not specified","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k."},{"observationId":"launch-mistral-1470","modelId":"qwen3.5-397b-a17b","label":"Not specified","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k."},{"observationId":"launch-mistral-1471","modelId":"mistral-medium-3.5","label":"Maximum reasoning","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k."},{"observationId":"launch-mistral-1472","modelId":"claude-sonnet-4-5-20250929","label":"Not specified","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k."},{"observationId":"launch-mistral-1473","modelId":"claude-sonnet-4-6","label":"Not specified","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k."},{"observationId":"launch-mistral-1474","modelId":"glm-5.1","label":"Not specified","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k."},{"observationId":"launch-mistral-1475","modelId":"qwen3.5-397b-a17b","label":"Not specified","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k."},{"observationId":"launch-mistral-1476","modelId":"mistral-medium-3.5","label":"Maximum reasoning","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k."},{"observationId":"launch-mistral-1477","modelId":"claude-sonnet-4-5-20250929","label":"Not specified","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k."},{"observationId":"launch-mistral-1478","modelId":"claude-sonnet-4-6","label":"Not specified","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k."},{"observationId":"launch-mistral-1479","modelId":"glm-5.1","label":"Not specified","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k."},{"observationId":"launch-mistral-1480","modelId":"qwen3.5-397b-a17b","label":"Not specified","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k."},{"observationId":"launch-mistral-1481","modelId":"mistral-medium-3.5","label":"Maximum reasoning","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k."},{"observationId":"launch-mistral-1482","modelId":"claude-sonnet-4-5-20250929","label":"Not specified","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k."},{"observationId":"launch-mistral-1483","modelId":"claude-sonnet-4-6","label":"Not specified","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k."},{"observationId":"launch-mistral-1484","modelId":"glm-5.1","label":"Not specified","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k."},{"observationId":"launch-mistral-1485","modelId":"qwen3.5-397b-a17b","label":"Not specified","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k."},{"observationId":"launch-mistral-1486","modelId":"mistral-medium-3.5","label":"Maximum reasoning","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k."},{"observationId":"launch-mistral-1487","modelId":"claude-sonnet-4-5-20250929","label":"Not specified","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k."},{"observationId":"launch-mistral-1488","modelId":"claude-sonnet-4-6","label":"Not specified","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k."},{"observationId":"launch-mistral-1489","modelId":"glm-5.1","label":"Not specified","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k."},{"observationId":"launch-mistral-1490","modelId":"qwen3.5-397b-a17b","label":"Not specified","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; tau3 4 trials, GPT5.2 low simulator except Sierra reported comparisons; BrowseComp discard-all context at100k."},{"observationId":"launch-mistral-1950","modelId":"magistral-medium-2509","label":"Maximum reasoning","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; Mistral same-lab comparison; tau3 4trials GPT5.2low user simulator; BrowseComp context-discard setup in card."},{"observationId":"launch-mistral-1951","modelId":"magistral-medium-2509","label":"Maximum reasoning","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; Mistral same-lab comparison; tau3 4trials GPT5.2low user simulator; BrowseComp context-discard setup in card."},{"observationId":"launch-mistral-1952","modelId":"magistral-medium-2509","label":"Maximum reasoning","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; Mistral same-lab comparison; tau3 4trials GPT5.2low user simulator; BrowseComp context-discard setup in card."},{"observationId":"launch-mistral-1953","modelId":"magistral-medium-2509","label":"Maximum reasoning","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; Mistral same-lab comparison; tau3 4trials GPT5.2low user simulator; BrowseComp context-discard setup in card."},{"observationId":"launch-mistral-1954","modelId":"magistral-medium-2509","label":"Maximum reasoning","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; Mistral same-lab comparison; tau3 4trials GPT5.2low user simulator; BrowseComp context-discard setup in card."},{"observationId":"launch-mistral-1955","modelId":"mistral-small-4","label":"Maximum reasoning","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; Mistral same-lab comparison; tau3 4trials GPT5.2low user simulator; BrowseComp context-discard setup in card."},{"observationId":"launch-mistral-1956","modelId":"mistral-small-4","label":"Maximum reasoning","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; Mistral same-lab comparison; tau3 4trials GPT5.2low user simulator; BrowseComp context-discard setup in card."},{"observationId":"launch-mistral-1957","modelId":"mistral-small-4","label":"Maximum reasoning","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; Mistral same-lab comparison; tau3 4trials GPT5.2low user simulator; BrowseComp context-discard setup in card."},{"observationId":"launch-mistral-1958","modelId":"mistral-small-4","label":"Maximum reasoning","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; Mistral same-lab comparison; tau3 4trials GPT5.2low user simulator; BrowseComp context-discard setup in card."},{"observationId":"launch-mistral-1959","modelId":"mistral-small-4","label":"Maximum reasoning","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; Mistral same-lab comparison; tau3 4trials GPT5.2low user simulator; BrowseComp context-discard setup in card."},{"observationId":"launch-mistral-1960","modelId":"mistral-medium-2508","label":"Maximum reasoning","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; Mistral same-lab comparison; tau3 4trials GPT5.2low user simulator; BrowseComp context-discard setup in card."},{"observationId":"launch-mistral-1961","modelId":"mistral-medium-2508","label":"Maximum reasoning","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; Mistral same-lab comparison; tau3 4trials GPT5.2low user simulator; BrowseComp context-discard setup in card."},{"observationId":"launch-mistral-1962","modelId":"mistral-medium-2508","label":"Maximum reasoning","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; Mistral same-lab comparison; tau3 4trials GPT5.2low user simulator; BrowseComp context-discard setup in card."},{"observationId":"launch-mistral-1963","modelId":"mistral-medium-2508","label":"Maximum reasoning","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; Mistral same-lab comparison; tau3 4trials GPT5.2low user simulator; BrowseComp context-discard setup in card."},{"observationId":"launch-mistral-1964","modelId":"mistral-medium-2508","label":"Maximum reasoning","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Maximum reasoning; Mistral same-lab comparison; tau3 4trials GPT5.2low user simulator; BrowseComp context-discard setup in card."},{"observationId":"launch-mistral-1965","modelId":"devstral-2512","label":"Not specified","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Mistral previous coding model comparison, published model-card harness settings."},{"observationId":"launch-mistral-1966","modelId":"labs-devstral-small-2512","label":"Not specified","sourceUrl":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configuration":"Mistral previous coding model comparison, published model-card harness settings."},{"observationId":"launch-nvidia-1000","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1001","modelId":"glm-5.1","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1002","modelId":"kimi-k2.6","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1003","modelId":"qwen3.5-397b-a17b","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1004","modelId":"nemotron-3-ultra","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1005","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1006","modelId":"glm-5.1","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1007","modelId":"kimi-k2.6","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1008","modelId":"qwen3.5-397b-a17b","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1009","modelId":"nemotron-3-ultra","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1010","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1011","modelId":"glm-5.1","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1012","modelId":"kimi-k2.6","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1013","modelId":"qwen3.5-397b-a17b","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1014","modelId":"nemotron-3-ultra","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1015","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1016","modelId":"glm-5.1","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1017","modelId":"kimi-k2.6","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1018","modelId":"qwen3.5-397b-a17b","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1019","modelId":"nemotron-3-ultra","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1020","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1021","modelId":"glm-5.1","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1022","modelId":"kimi-k2.6","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1023","modelId":"qwen3.5-397b-a17b","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1024","modelId":"nemotron-3-ultra","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1025","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1026","modelId":"glm-5.1","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1027","modelId":"kimi-k2.6","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1028","modelId":"qwen3.5-397b-a17b","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1029","modelId":"nemotron-3-ultra","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1030","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1031","modelId":"glm-5.1","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1032","modelId":"kimi-k2.6","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1033","modelId":"qwen3.5-397b-a17b","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1034","modelId":"nemotron-3-ultra","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1035","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1036","modelId":"glm-5.1","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1037","modelId":"kimi-k2.6","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1038","modelId":"qwen3.5-397b-a17b","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1039","modelId":"nemotron-3-ultra","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1040","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1041","modelId":"glm-5.1","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1042","modelId":"kimi-k2.6","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1043","modelId":"qwen3.5-397b-a17b","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1044","modelId":"nemotron-3-ultra","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1045","modelId":"glm-5.1","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1046","modelId":"kimi-k2.6","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1047","modelId":"qwen3.5-397b-a17b","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1048","modelId":"nemotron-3-ultra","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1049","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1050","modelId":"glm-5.1","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1051","modelId":"kimi-k2.6","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1052","modelId":"qwen3.5-397b-a17b","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1053","modelId":"nemotron-3-ultra","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1054","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1055","modelId":"glm-5.1","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1056","modelId":"kimi-k2.6","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1057","modelId":"qwen3.5-397b-a17b","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1058","modelId":"nemotron-3-ultra","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1059","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1060","modelId":"glm-5.1","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1061","modelId":"kimi-k2.6","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1062","modelId":"qwen3.5-397b-a17b","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1063","modelId":"nemotron-3-ultra","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1064","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1065","modelId":"glm-5.1","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1066","modelId":"kimi-k2.6","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1067","modelId":"qwen3.5-397b-a17b","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1068","modelId":"nemotron-3-ultra","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1069","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1070","modelId":"glm-5.1","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1071","modelId":"kimi-k2.6","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1072","modelId":"qwen3.5-397b-a17b","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1073","modelId":"nemotron-3-ultra","label":"Thinking","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1074","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1075","modelId":"glm-5.1","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1076","modelId":"kimi-k2.6","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1077","modelId":"qwen3.5-397b-a17b","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1078","modelId":"nemotron-3-ultra","label":"Thinking","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1079","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1080","modelId":"glm-5.1","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1081","modelId":"kimi-k2.6","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1082","modelId":"qwen3.5-397b-a17b","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1083","modelId":"nemotron-3-ultra","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1084","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1085","modelId":"glm-5.1","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1086","modelId":"kimi-k2.6","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1087","modelId":"qwen3.5-397b-a17b","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1088","modelId":"nemotron-3-ultra","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1089","modelId":"glm-5.1","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1090","modelId":"kimi-k2.6","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1091","modelId":"qwen3.5-397b-a17b","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1092","modelId":"nemotron-3-ultra","label":"Thinking","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1093","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1094","modelId":"glm-5.1","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1095","modelId":"kimi-k2.6","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1096","modelId":"qwen3.5-397b-a17b","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1097","modelId":"nemotron-3-ultra","label":"Thinking","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1098","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1099","modelId":"glm-5.1","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1100","modelId":"kimi-k2.6","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1101","modelId":"qwen3.5-397b-a17b","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1102","modelId":"nemotron-3-ultra","label":"Thinking","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1103","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1104","modelId":"glm-5.1","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1105","modelId":"kimi-k2.6","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1106","modelId":"qwen3.5-397b-a17b","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1107","modelId":"nemotron-3-ultra","label":"Thinking","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1108","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1109","modelId":"glm-5.1","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1110","modelId":"kimi-k2.6","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1111","modelId":"qwen3.5-397b-a17b","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1112","modelId":"nemotron-3-ultra","label":"Thinking","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1113","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1114","modelId":"glm-5.1","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1115","modelId":"kimi-k2.6","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1116","modelId":"qwen3.5-397b-a17b","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1117","modelId":"nemotron-3-ultra","label":"Thinking","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1118","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1119","modelId":"glm-5.1","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1120","modelId":"kimi-k2.6","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1121","modelId":"qwen3.5-397b-a17b","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1122","modelId":"nemotron-3-ultra","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1123","modelId":"qwen3.5-397b-a17b","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1124","modelId":"nemotron-3-ultra","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1125","modelId":"qwen3.5-397b-a17b","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1126","modelId":"nemotron-3-ultra","label":"Thinking","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1127","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1128","modelId":"glm-5.1","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1129","modelId":"kimi-k2.6","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1130","modelId":"qwen3.5-397b-a17b","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1131","modelId":"nemotron-3-ultra","label":"Thinking","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1132","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1133","modelId":"glm-5.1","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1134","modelId":"kimi-k2.6","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-1135","modelId":"qwen3.5-397b-a17b","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-974","modelId":"nemotron-3-ultra","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-975","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-976","modelId":"glm-5.1","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-977","modelId":"kimi-k2.6","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-978","modelId":"qwen3.5-397b-a17b","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-979","modelId":"nemotron-3-ultra","label":"Thinking","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-980","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-981","modelId":"glm-5.1","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-982","modelId":"kimi-k2.6","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-983","modelId":"qwen3.5-397b-a17b","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-984","modelId":"nemotron-3-ultra","label":"Thinking","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-985","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-986","modelId":"glm-5.1","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-987","modelId":"kimi-k2.6","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-988","modelId":"qwen3.5-397b-a17b","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-989","modelId":"nemotron-3-ultra","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-990","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-991","modelId":"glm-5.1","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-992","modelId":"kimi-k2.6","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-993","modelId":"qwen3.5-397b-a17b","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-994","modelId":"nemotron-3-ultra","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-995","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-996","modelId":"glm-5.1","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-997","modelId":"kimi-k2.6","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-998","modelId":"qwen3.5-397b-a17b","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-nvidia-999","modelId":"nemotron-3-ultra","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configuration":"NVIDIA evaluation harness/settings per benchmark in model card; comparator scores are NVIDIA-reported under that protocol."},{"observationId":"launch-openai-astra-launch-137","modelId":"gpt-6-astra","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-138","modelId":"gpt-5.6-sol","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-139","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-140","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-141","modelId":"gpt-6-astra","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; offline set; partial credit; v2026.08.08; official task/grading settings"},{"observationId":"launch-openai-astra-launch-142","modelId":"gpt-5.6-sol","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; offline set; partial credit; v2026.08.08; official task/grading settings"},{"observationId":"launch-openai-astra-launch-143","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; offline set; partial credit; v2026.08.08; official task/grading settings"},{"observationId":"launch-openai-astra-launch-144","modelId":"gpt-6-astra","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; no tools"},{"observationId":"launch-openai-astra-launch-145","modelId":"gpt-5.6-sol","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; no tools"},{"observationId":"launch-openai-astra-launch-146","modelId":"gpt-6-astra","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-147","modelId":"gpt-5.6-sol","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-148","modelId":"claude-fable-5-1","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-149","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-150","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-151","modelId":"gpt-6-astra","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; tools enabled"},{"observationId":"launch-openai-astra-launch-152","modelId":"gpt-5.6-sol","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; tools enabled"},{"observationId":"launch-openai-astra-launch-153","modelId":"claude-fable-5-1","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; tools enabled; three Anthropic evaluation modifications"},{"observationId":"launch-openai-astra-launch-154","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; tools enabled; three Anthropic evaluation modifications"},{"observationId":"launch-openai-astra-launch-155","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; tools enabled; three Anthropic evaluation modifications"},{"observationId":"launch-openai-astra-launch-156","modelId":"gpt-6-astra","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-157","modelId":"gpt-5.6-sol","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-158","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-159","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-160","modelId":"gpt-6-astra","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-161","modelId":"gpt-5.6-sol","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-162","modelId":"gpt-6-astra","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-163","modelId":"gpt-5.6-sol","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-164","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-165","modelId":"gpt-6-astra","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-166","modelId":"gpt-5.6-sol","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-167","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-168","modelId":"gpt-6-astra","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-169","modelId":"gpt-5.6-sol","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-170","modelId":"claude-fable-5-1","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-171","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-172","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-173","modelId":"gemini-3.8-flash","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-174","modelId":"gpt-6-astra","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-175","modelId":"gpt-5.6-sol","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-176","modelId":"claude-fable-5-1","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-177","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-178","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-179","modelId":"gemini-3.8-flash","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-180","modelId":"gpt-6-astra","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-181","modelId":"gpt-5.6-sol","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-182","modelId":"claude-fable-5-1","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-183","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-184","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-185","modelId":"gemini-3.8-flash","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-186","modelId":"gpt-6-astra","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; Codex-style developer instruction on tests, reuse and repository conventions"},{"observationId":"launch-openai-astra-launch-187","modelId":"gpt-5.6-sol","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-188","modelId":"claude-fable-5-1","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-189","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-190","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-191","modelId":"gemini-3.8-flash","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-192","modelId":"gpt-6-astra","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; Codex-style developer instruction on tests, reuse and repository conventions"},{"observationId":"launch-openai-astra-launch-193","modelId":"gpt-5.6-sol","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-194","modelId":"claude-fable-5-1","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-195","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-196","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-197","modelId":"gemini-3.8-flash","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-198","modelId":"gpt-6-astra","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-199","modelId":"gpt-5.6-sol","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-200","modelId":"claude-fable-5-1","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-201","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-202","modelId":"gpt-6-astra","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-203","modelId":"gpt-5.6-sol","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-204","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-205","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-206","modelId":"gemini-3.8-flash","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-207","modelId":"gpt-6-astra","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-208","modelId":"gpt-5.6-sol","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-209","modelId":"claude-fable-5-1","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-210","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-211","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-212","modelId":"gpt-6-astra","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-213","modelId":"gpt-5.6-sol","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-214","modelId":"claude-fable-5-1","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-215","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-216","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-217","modelId":"gpt-6-astra","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-218","modelId":"gpt-5.6-sol","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-219","modelId":"claude-fable-5-1","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-220","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-221","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-222","modelId":"gemini-3.8-flash","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-223","modelId":"gpt-6-astra","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; tools enabled"},{"observationId":"launch-openai-astra-launch-224","modelId":"claude-fable-5-1","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; tools enabled"},{"observationId":"launch-openai-astra-launch-225","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; tools enabled"},{"observationId":"launch-openai-astra-launch-226","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; tools enabled"},{"observationId":"launch-openai-astra-launch-227","modelId":"gpt-6-astra","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-228","modelId":"gpt-5.6-sol","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-229","modelId":"gpt-6-astra","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-230","modelId":"gpt-5.6-sol","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-231","modelId":"gpt-6-astra","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-232","modelId":"gpt-5.6-sol","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-233","modelId":"gpt-6-astra","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; official paper scoring; length-adjusted"},{"observationId":"launch-openai-astra-launch-234","modelId":"gpt-5.6-sol","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; official paper scoring; length-adjusted"},{"observationId":"launch-openai-astra-launch-235","modelId":"claude-fable-5-1","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; official paper scoring; length-adjusted; OpenAI reproduction; GPT-5.4 grader; unclipped; Opus5 fallback for provider refusals"},{"observationId":"launch-openai-astra-launch-236","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; official paper scoring; length-adjusted; OpenAI reproduction; GPT-5.4 grader; unclipped"},{"observationId":"launch-openai-astra-launch-237","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; official paper scoring; length-adjusted; OpenAI reproduction; GPT-5.4 grader; unclipped"},{"observationId":"launch-openai-astra-launch-238","modelId":"gemini-3.8-flash","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; official paper scoring; length-adjusted"},{"observationId":"launch-openai-astra-launch-239","modelId":"gpt-6-astra","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; reduced or absent production safeguards"},{"observationId":"launch-openai-astra-launch-240","modelId":"gpt-5.6-sol","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; reduced or absent production safeguards"},{"observationId":"launch-openai-astra-launch-241","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; reduced or absent production safeguards"},{"observationId":"launch-openai-astra-launch-242","modelId":"gpt-6-astra","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; reduced or absent production safeguards; v1 offline environment; no runtime package installation; token capped, no wall-clock cap"},{"observationId":"launch-openai-astra-launch-243","modelId":"gpt-5.6-sol","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; reduced or absent production safeguards; v1 offline environment; no runtime package installation; token capped, no wall-clock cap"},{"observationId":"launch-openai-astra-launch-244","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; reduced or absent production safeguards; launch-reported configuration"},{"observationId":"launch-openai-astra-launch-245","modelId":"gpt-6-astra","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; reduced or absent production safeguards; 20 vulnerabilities / 13 Chrome releases"},{"observationId":"launch-openai-astra-launch-246","modelId":"gpt-5.6-sol","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; reduced or absent production safeguards; 20 vulnerabilities / 13 Chrome releases; 300-turn limit"},{"observationId":"launch-openai-astra-launch-247","modelId":"gpt-6-astra","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; reduced or absent production safeguards; pass@1; all six objectives required"},{"observationId":"launch-openai-astra-launch-248","modelId":"gpt-5.6-sol","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; reduced or absent production safeguards; pass@1; all six objectives required"},{"observationId":"launch-openai-astra-launch-249","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; reduced or absent production safeguards; pass@1; all six objectives required"},{"observationId":"launch-openai-astra-launch-250","modelId":"gpt-6-astra","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; reduced or absent production safeguards; May2026 JavaScript subset,183 vulnerabilities; revised agent root-cause grader"},{"observationId":"launch-openai-astra-launch-251","modelId":"gpt-5.6-sol","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; reduced or absent production safeguards; May2026 JavaScript subset,183 vulnerabilities; revised agent root-cause grader"},{"observationId":"launch-openai-astra-launch-252","modelId":"gpt-6-astra","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; 8-needle 256K-512K"},{"observationId":"launch-openai-astra-launch-253","modelId":"gpt-5.6-sol","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; 8-needle 256K-512K"},{"observationId":"launch-openai-astra-launch-254","modelId":"gpt-6-astra","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; 8-needle 512K-1M"},{"observationId":"launch-openai-astra-launch-255","modelId":"gpt-5.6-sol","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; 8-needle 512K-1M"},{"observationId":"launch-openai-astra-launch-256","modelId":"gpt-6-astra","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API; Responses API harness with two documented setting changes"},{"observationId":"launch-openai-astra-launch-257","modelId":"gpt-5.6-sol","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-258","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-259","modelId":"gpt-6-astra","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-260","modelId":"gpt-5.6-sol","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-261","modelId":"claude-fable-5-1","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-262","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-263","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-264","modelId":"gpt-6-astra","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-265","modelId":"gpt-5.6-sol","label":"Best across efforts","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-266","modelId":"claude-fable-5-1","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-267","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-268","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Maximum reported across reasoning efforts; OpenAI research environment or API"},{"observationId":"launch-openai-astra-launch-547","modelId":"gpt-6-astra","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Lower-cost setting; exact effort not specified"},{"observationId":"launch-openai-astra-launch-548","modelId":"gpt-6-astra","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Lower-cost setting; exact effort not specified"},{"observationId":"launch-openai-astra-launch-549","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-6-astra/","configuration":"Similar settings with fewer300-turn-limit interruptions; production safeguards absent"},{"observationId":"launch-openai-astra-system-card-489","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"length-adjusted; official HealthBench scoring"},{"observationId":"launch-openai-astra-system-card-490","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"unadjusted; official HealthBench scoring"},{"observationId":"launch-openai-astra-system-card-491","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"length-adjusted; official HealthBench scoring"},{"observationId":"launch-openai-astra-system-card-492","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"unadjusted; official HealthBench scoring"},{"observationId":"launch-openai-astra-system-card-493","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"length-adjusted; official HealthBench scoring"},{"observationId":"launch-openai-astra-system-card-494","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"unadjusted; official HealthBench scoring"},{"observationId":"launch-openai-astra-system-card-495","modelId":"gpt-5.6-luna","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"length-adjusted; official HealthBench scoring"},{"observationId":"launch-openai-astra-system-card-496","modelId":"gpt-5.6-luna","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"unadjusted; official HealthBench scoring"},{"observationId":"launch-openai-astra-system-card-497","modelId":"gpt-6-astra","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"length-adjusted; official HealthBench scoring"},{"observationId":"launch-openai-astra-system-card-498","modelId":"gpt-6-astra","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"unadjusted; official HealthBench scoring"},{"observationId":"launch-openai-astra-system-card-499","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"length-adjusted; official HealthBench scoring"},{"observationId":"launch-openai-astra-system-card-500","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"unadjusted; official HealthBench scoring"},{"observationId":"launch-openai-astra-system-card-501","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"length-adjusted; official HealthBench scoring"},{"observationId":"launch-openai-astra-system-card-502","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"unadjusted; official HealthBench scoring"},{"observationId":"launch-openai-astra-system-card-503","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"length-adjusted; official HealthBench scoring"},{"observationId":"launch-openai-astra-system-card-504","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"unadjusted; official HealthBench scoring"},{"observationId":"launch-openai-astra-system-card-505","modelId":"gpt-5.6-luna","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"length-adjusted; official HealthBench scoring"},{"observationId":"launch-openai-astra-system-card-506","modelId":"gpt-5.6-luna","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"unadjusted; official HealthBench scoring"},{"observationId":"launch-openai-astra-system-card-507","modelId":"gpt-6-astra","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"length-adjusted; official HealthBench scoring"},{"observationId":"launch-openai-astra-system-card-508","modelId":"gpt-6-astra","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"unadjusted; official HealthBench scoring"},{"observationId":"launch-openai-astra-system-card-509","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"length-adjusted; official HealthBench scoring"},{"observationId":"launch-openai-astra-system-card-510","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"unadjusted; official HealthBench scoring"},{"observationId":"launch-openai-astra-system-card-511","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"length-adjusted; official HealthBench scoring"},{"observationId":"launch-openai-astra-system-card-512","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"unadjusted; official HealthBench scoring"},{"observationId":"launch-openai-astra-system-card-513","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"length-adjusted; official HealthBench scoring"},{"observationId":"launch-openai-astra-system-card-514","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"unadjusted; official HealthBench scoring"},{"observationId":"launch-openai-astra-system-card-515","modelId":"gpt-5.6-luna","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"length-adjusted; official HealthBench scoring"},{"observationId":"launch-openai-astra-system-card-516","modelId":"gpt-5.6-luna","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"unadjusted; official HealthBench scoring"},{"observationId":"launch-openai-astra-system-card-517","modelId":"gpt-6-astra","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"length-adjusted; official HealthBench scoring"},{"observationId":"launch-openai-astra-system-card-518","modelId":"gpt-6-astra","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"unadjusted; official HealthBench scoring"},{"observationId":"launch-openai-astra-system-card-519","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"length-adjusted; official HealthBench scoring"},{"observationId":"launch-openai-astra-system-card-520","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"unadjusted; official HealthBench scoring"},{"observationId":"launch-openai-astra-system-card-521","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"length-adjusted; official HealthBench scoring"},{"observationId":"launch-openai-astra-system-card-522","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"unadjusted; official HealthBench scoring"},{"observationId":"launch-openai-astra-system-card-523","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"length-adjusted; official HealthBench scoring"},{"observationId":"launch-openai-astra-system-card-524","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"unadjusted; official HealthBench scoring"},{"observationId":"launch-openai-astra-system-card-525","modelId":"gpt-5.6-luna","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"length-adjusted; official HealthBench scoring"},{"observationId":"launch-openai-astra-system-card-526","modelId":"gpt-5.6-luna","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"unadjusted; official HealthBench scoring"},{"observationId":"launch-openai-astra-system-card-527","modelId":"gpt-6-astra","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"length-adjusted; official HealthBench scoring"},{"observationId":"launch-openai-astra-system-card-528","modelId":"gpt-6-astra","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"unadjusted; official HealthBench scoring"},{"observationId":"launch-openai-astra-system-card-529","modelId":"gpt-6-astra","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"System-card reported configuration; reasoning effort unspecified"},{"observationId":"launch-openai-astra-system-card-530","modelId":"gpt-6-astra","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"pass@4; four independent trials; all six objectives required; reduced production safeguards"},{"observationId":"launch-openai-astra-system-card-531","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"pass@4; four independent trials; all six objectives required; reduced production safeguards"},{"observationId":"launch-openai-astra-system-card-532","modelId":"gpt-6-astra","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"22 isolated CTF-style targets; protected-flag success metric"},{"observationId":"launch-openai-astra-system-card-533","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"22 isolated CTF-style targets; protected-flag success metric"},{"observationId":"launch-openai-astra-system-card-534","modelId":"gpt-6-astra","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"UK AISI; single forward pass; no chain of thought"},{"observationId":"launch-openai-astra-system-card-535","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"UK AISI; single forward pass; no chain of thought"},{"observationId":"launch-openai-astra-system-card-536","modelId":"gpt-6-astra","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"observed performance"},{"observationId":"launch-openai-astra-system-card-537","modelId":"gpt-6-astra","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"refusal-adjusted upper estimate; refusals counted as successes"},{"observationId":"launch-openai-astra-system-card-538","modelId":"gpt-6-astra","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"observed performance"},{"observationId":"launch-openai-astra-system-card-539","modelId":"gpt-6-astra","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"refusal-adjusted upper estimate; refusals counted as successes"},{"observationId":"launch-openai-astra-system-card-540","modelId":"gpt-6-astra","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"observed performance"},{"observationId":"launch-openai-astra-system-card-541","modelId":"gpt-6-astra","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"refusal-adjusted upper estimate; refusals counted as successes"},{"observationId":"launch-openai-astra-system-card-542","modelId":"gpt-6-astra","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"System-card reported configuration; reasoning effort unspecified"},{"observationId":"launch-openai-astra-system-card-543","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"System-card reported configuration; reasoning effort unspecified"},{"observationId":"launch-openai-astra-system-card-544","modelId":"gpt-6-astra","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"System-card reported configuration; reasoning effort unspecified"},{"observationId":"launch-openai-astra-system-card-545","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"System-card reported configuration; reasoning effort unspecified"},{"observationId":"launch-openai-astra-system-card-546","modelId":"gpt-6-astra","label":"Not specified","sourceUrl":"https://deploymentsafety.openai.com/gpt-6-astra","configuration":"System-card reported configuration; reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-269","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-270","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-271","modelId":"gpt-5.6-luna","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-272","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-273","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-274","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-275","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-276","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-277","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-278","modelId":"gpt-5.6-luna","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-279","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-280","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-281","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-282","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-283","modelId":"gemini-3.5-flash","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-284","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-285","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-286","modelId":"gpt-5.6-luna","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-287","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-288","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-289","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-290","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-291","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-292","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-293","modelId":"gpt-5.6-luna","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-294","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-295","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-296","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-297","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-298","modelId":"gpt-5.6-luna","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-299","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-300","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-301","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-302","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-303","modelId":"gemini-3.5-flash","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-304","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-305","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-306","modelId":"gpt-5.6-luna","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-307","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-308","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-309","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-310","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-311","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-312","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-313","modelId":"gpt-5.6-luna","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-314","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-315","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-316","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-317","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-318","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-319","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-320","modelId":"gpt-5.6-luna","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-321","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-322","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-323","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-324","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-325","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-326","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; Ultra, four-agent orchestration"},{"observationId":"launch-openai-sol-launch-327","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-328","modelId":"gpt-5.6-luna","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-329","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-330","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-331","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-332","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-333","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-334","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-335","modelId":"gpt-5.6-luna","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-336","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-337","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-338","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-339","modelId":"gemini-3.5-flash","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-340","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-341","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-342","modelId":"gpt-5.6-luna","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-343","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-344","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-345","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-346","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-347","modelId":"gpt-5.6-luna","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-348","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-349","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; official paper scoring; length-adjusted"},{"observationId":"launch-openai-sol-launch-350","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; official paper scoring; length-adjusted"},{"observationId":"launch-openai-sol-launch-351","modelId":"gpt-5.6-luna","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; official paper scoring; length-adjusted"},{"observationId":"launch-openai-sol-launch-352","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; official paper scoring; length-adjusted"},{"observationId":"launch-openai-sol-launch-353","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; official paper scoring; length-adjusted"},{"observationId":"launch-openai-sol-launch-354","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; official paper scoring; length-adjusted"},{"observationId":"launch-openai-sol-launch-355","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-356","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-357","modelId":"gpt-5.6-luna","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-358","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-359","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-360","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-361","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; Ultra, four-agent orchestration"},{"observationId":"launch-openai-sol-launch-362","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-363","modelId":"gpt-5.6-luna","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-364","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-365","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-366","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-367","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-368","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-369","modelId":"gpt-5.6-luna","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-370","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-371","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-372","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; Python tool enabled"},{"observationId":"launch-openai-sol-launch-373","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; Python tool enabled"},{"observationId":"launch-openai-sol-launch-374","modelId":"gpt-5.6-luna","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; Python tool enabled"},{"observationId":"launch-openai-sol-launch-375","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; Python tool enabled"},{"observationId":"launch-openai-sol-launch-376","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; Python tool enabled"},{"observationId":"launch-openai-sol-launch-377","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; reduced or absent production safeguards"},{"observationId":"launch-openai-sol-launch-378","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; reduced or absent production safeguards"},{"observationId":"launch-openai-sol-launch-379","modelId":"gpt-5.6-luna","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; reduced or absent production safeguards"},{"observationId":"launch-openai-sol-launch-380","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; reduced or absent production safeguards"},{"observationId":"launch-openai-sol-launch-381","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; reduced or absent production safeguards; public grader; May2026 JavaScript subset"},{"observationId":"launch-openai-sol-launch-382","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; Ultra, four-agent orchestration; reduced or absent production safeguards; public grader; May2026 JavaScript subset"},{"observationId":"launch-openai-sol-launch-383","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; reduced or absent production safeguards; public grader; May2026 JavaScript subset"},{"observationId":"launch-openai-sol-launch-384","modelId":"gpt-5.6-luna","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; reduced or absent production safeguards; public grader; May2026 JavaScript subset"},{"observationId":"launch-openai-sol-launch-385","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; reduced or absent production safeguards; public grader; May2026 JavaScript subset"},{"observationId":"launch-openai-sol-launch-386","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; reduced or absent production safeguards; ExploitBench API harness; five seeds; reasoning continuity"},{"observationId":"launch-openai-sol-launch-387","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; reduced or absent production safeguards; ExploitBench API harness; five seeds; reasoning continuity"},{"observationId":"launch-openai-sol-launch-388","modelId":"gpt-5.6-luna","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; reduced or absent production safeguards; ExploitBench API harness; five seeds; reasoning continuity"},{"observationId":"launch-openai-sol-launch-389","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; reduced or absent production safeguards; ExploitBench API harness; five seeds; reasoning continuity"},{"observationId":"launch-openai-sol-launch-390","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; reduced or absent production safeguards; ExploitBench API harness; five seeds; reasoning continuity"},{"observationId":"launch-openai-sol-launch-391","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; reduced or absent production safeguards; six-hour evaluation cap; alpha API latency rescaled to public API"},{"observationId":"launch-openai-sol-launch-392","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; reduced or absent production safeguards; six-hour evaluation cap; alpha API latency rescaled to public API"},{"observationId":"launch-openai-sol-launch-393","modelId":"gpt-5.6-luna","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; reduced or absent production safeguards; six-hour evaluation cap; alpha API latency rescaled to public API"},{"observationId":"launch-openai-sol-launch-394","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; reduced or absent production safeguards; six-hour evaluation cap; alpha API latency rescaled to public API"},{"observationId":"launch-openai-sol-launch-395","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-396","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-397","modelId":"gpt-5.6-luna","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-398","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-399","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-400","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-401","modelId":"gpt-5.6-luna","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-402","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-403","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-404","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-405","modelId":"gpt-5.6-luna","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-406","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-407","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-408","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-409","modelId":"gpt-5.6-luna","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-410","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-411","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-412","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-413","modelId":"gpt-5.6-luna","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-414","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-415","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; no tools"},{"observationId":"launch-openai-sol-launch-416","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; no tools"},{"observationId":"launch-openai-sol-launch-417","modelId":"gpt-5.6-luna","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; no tools"},{"observationId":"launch-openai-sol-launch-418","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; no tools"},{"observationId":"launch-openai-sol-launch-419","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; no tools"},{"observationId":"launch-openai-sol-launch-420","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; tools enabled"},{"observationId":"launch-openai-sol-launch-421","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; tools enabled"},{"observationId":"launch-openai-sol-launch-422","modelId":"gpt-5.6-luna","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; tools enabled"},{"observationId":"launch-openai-sol-launch-423","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; tools enabled"},{"observationId":"launch-openai-sol-launch-424","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-425","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-426","modelId":"gpt-5.6-luna","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-427","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-428","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-429","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-430","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-431","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-432","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-433","modelId":"gpt-5.6-luna","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-434","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-435","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-436","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-437","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-438","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-439","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-440","modelId":"gpt-5.6-luna","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-441","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-442","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-443","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-444","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-445","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-446","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-447","modelId":"gpt-5.6-luna","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-448","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-449","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-450","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-451","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-452","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-453","modelId":"gpt-5.6-luna","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-454","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-455","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-456","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-457","modelId":"gemini-3.5-flash","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-458","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-459","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-460","modelId":"gpt-5.6-luna","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-461","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-462","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-463","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-464","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-465","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; 8-needle 256K-512K"},{"observationId":"launch-openai-sol-launch-466","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; 8-needle 256K-512K"},{"observationId":"launch-openai-sol-launch-467","modelId":"gpt-5.6-luna","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; 8-needle 256K-512K"},{"observationId":"launch-openai-sol-launch-468","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; 8-needle 256K-512K"},{"observationId":"launch-openai-sol-launch-469","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; 8-needle 512K-1M"},{"observationId":"launch-openai-sol-launch-470","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; 8-needle 512K-1M"},{"observationId":"launch-openai-sol-launch-471","modelId":"gpt-5.6-luna","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; 8-needle 512K-1M"},{"observationId":"launch-openai-sol-launch-472","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; 8-needle 512K-1M"},{"observationId":"launch-openai-sol-launch-473","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; BFS 256k f1"},{"observationId":"launch-openai-sol-launch-474","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; BFS 256k f1"},{"observationId":"launch-openai-sol-launch-475","modelId":"gpt-5.6-luna","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; BFS 256k f1"},{"observationId":"launch-openai-sol-launch-476","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; BFS 256k f1"},{"observationId":"launch-openai-sol-launch-477","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; BFS 256k f1"},{"observationId":"launch-openai-sol-launch-478","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; BFS 1mil f1"},{"observationId":"launch-openai-sol-launch-479","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; BFS 1mil f1"},{"observationId":"launch-openai-sol-launch-480","modelId":"gpt-5.6-luna","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; BFS 1mil f1"},{"observationId":"launch-openai-sol-launch-481","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; BFS 1mil f1"},{"observationId":"launch-openai-sol-launch-482","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; BFS 1mil f1"},{"observationId":"launch-openai-sol-launch-483","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-484","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-485","modelId":"gpt-5.6-luna","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-486","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-487","modelId":"claude-opus-4-8","label":"High","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified; high reasoning, not max"},{"observationId":"launch-openai-sol-launch-488","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Launch table reported configuration; per-cell reasoning effort unspecified"},{"observationId":"launch-openai-sol-launch-550","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":"Two-hour cap; alpha API latency rescaled; reduced safeguards"},{"observationId":"launch-qwen-551","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Qwen Claude Code avg@10,5h timeout,131072 output; comparators best published across harnesses."},{"observationId":"launch-qwen-552","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Qwen Claude Code avg@10,5h timeout,131072 output; comparators best published across harnesses."},{"observationId":"launch-qwen-553","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Qwen Claude Code avg@10,5h timeout,131072 output; comparators best published across harnesses."},{"observationId":"launch-qwen-554","modelId":"qwen3.7-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Qwen Claude Code avg@10,5h timeout,131072 output; comparators best published across harnesses."},{"observationId":"launch-qwen-555","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Qwen Claude Code avg@10,5h timeout,131072 output; comparators best published across harnesses."},{"observationId":"launch-qwen-556","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Refined task set with problematic tasks corrected; all baselines rerun, Claude Code,temp1,top_p.95,256K context."},{"observationId":"launch-qwen-557","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Refined task set with problematic tasks corrected; all baselines rerun, Claude Code,temp1,top_p.95,256K context."},{"observationId":"launch-qwen-558","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Refined task set with problematic tasks corrected; all baselines rerun, Claude Code,temp1,top_p.95,256K context."},{"observationId":"launch-qwen-559","modelId":"qwen3.7-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Refined task set with problematic tasks corrected; all baselines rerun, Claude Code,temp1,top_p.95,256K context."},{"observationId":"launch-qwen-560","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Refined task set with problematic tasks corrected; all baselines rerun, Claude Code,temp1,top_p.95,256K context."},{"observationId":"launch-qwen-561","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Best of Claude Code and mini-SWE-agent; Qwen best Claude Code; temp1,top_p.95,256K."},{"observationId":"launch-qwen-562","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Best of Claude Code and mini-SWE-agent; Qwen best Claude Code; temp1,top_p.95,256K."},{"observationId":"launch-qwen-563","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Best of Claude Code and mini-SWE-agent; Qwen best Claude Code; temp1,top_p.95,256K."},{"observationId":"launch-qwen-564","modelId":"qwen3.7-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Best of Claude Code and mini-SWE-agent; Qwen best Claude Code; temp1,top_p.95,256K."},{"observationId":"launch-qwen-565","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Best of Claude Code and mini-SWE-agent; Qwen best Claude Code; temp1,top_p.95,256K."},{"observationId":"launch-qwen-566","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-567","modelId":"qwen3.7-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-568","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-569","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-570","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-571","modelId":"qwen3.7-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-572","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-573","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-574","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-575","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-576","modelId":"qwen3.7-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-577","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-578","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. BasicAgent Code-Dev; Opus4.6 judge,3runs,12h each."},{"observationId":"launch-qwen-579","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. BasicAgent Code-Dev; Opus4.6 judge,3runs,12h each."},{"observationId":"launch-qwen-580","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. BasicAgent Code-Dev; Opus4.6 judge,3runs,12h each."},{"observationId":"launch-qwen-581","modelId":"qwen3.7-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. BasicAgent Code-Dev; Opus4.6 judge,3runs,12h each."},{"observationId":"launch-qwen-582","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. BasicAgent Code-Dev; Opus4.6 judge,3runs,12h each."},{"observationId":"launch-qwen-583","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-584","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-585","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-586","modelId":"qwen3.7-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-587","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-588","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal software engineering,ClaudeCode,avg@3,8h,32768 output,temp1,256K."},{"observationId":"launch-qwen-589","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal software engineering,ClaudeCode,avg@3,8h,32768 output,temp1,256K."},{"observationId":"launch-qwen-590","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal software engineering,ClaudeCode,avg@3,8h,32768 output,temp1,256K."},{"observationId":"launch-qwen-591","modelId":"qwen3.7-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal software engineering,ClaudeCode,avg@3,8h,32768 output,temp1,256K."},{"observationId":"launch-qwen-592","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal software engineering,ClaudeCode,avg@3,8h,32768 output,temp1,256K."},{"observationId":"launch-qwen-593","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal Qoder tasks,ClaudeCode,avg@5,6h,32768 output,temp1,256K."},{"observationId":"launch-qwen-594","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal Qoder tasks,ClaudeCode,avg@5,6h,32768 output,temp1,256K."},{"observationId":"launch-qwen-595","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal Qoder tasks,ClaudeCode,avg@5,6h,32768 output,temp1,256K."},{"observationId":"launch-qwen-596","modelId":"qwen3.7-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal Qoder tasks,ClaudeCode,avg@5,6h,32768 output,temp1,256K."},{"observationId":"launch-qwen-597","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal Qoder tasks,ClaudeCode,avg@5,6h,32768 output,temp1,256K."},{"observationId":"launch-qwen-598","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal bilingual EN/CN React benchmark,7categories,ClaudeCode,render+multimodaljudge,BT/Elo."},{"observationId":"launch-qwen-599","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal bilingual EN/CN React benchmark,7categories,ClaudeCode,render+multimodaljudge,BT/Elo."},{"observationId":"launch-qwen-600","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal bilingual EN/CN React benchmark,7categories,ClaudeCode,render+multimodaljudge,BT/Elo."},{"observationId":"launch-qwen-601","modelId":"qwen3.7-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal bilingual EN/CN React benchmark,7categories,ClaudeCode,render+multimodaljudge,BT/Elo."},{"observationId":"launch-qwen-602","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal bilingual EN/CN React benchmark,7categories,ClaudeCode,render+multimodaljudge,BT/Elo."},{"observationId":"launch-qwen-603","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal bilingual EN/CN SVG benchmark,render+multimodaljudge,BT/Elo."},{"observationId":"launch-qwen-604","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal bilingual EN/CN SVG benchmark,render+multimodaljudge,BT/Elo."},{"observationId":"launch-qwen-605","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal bilingual EN/CN SVG benchmark,render+multimodaljudge,BT/Elo."},{"observationId":"launch-qwen-606","modelId":"qwen3.7-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal bilingual EN/CN SVG benchmark,render+multimodaljudge,BT/Elo."},{"observationId":"launch-qwen-607","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal bilingual EN/CN SVG benchmark,render+multimodaljudge,BT/Elo."},{"observationId":"launch-qwen-608","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal professional work tasks across science,finance,law,medical,productivity."},{"observationId":"launch-qwen-609","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal professional work tasks across science,finance,law,medical,productivity."},{"observationId":"launch-qwen-610","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal professional work tasks across science,finance,law,medical,productivity."},{"observationId":"launch-qwen-611","modelId":"qwen3.7-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal professional work tasks across science,finance,law,medical,productivity."},{"observationId":"launch-qwen-612","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Internal professional work tasks across science,finance,law,medical,productivity."},{"observationId":"launch-qwen-613","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-614","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-615","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-616","modelId":"qwen3.7-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-617","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-618","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-619","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-620","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-621","modelId":"qwen3.7-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-622","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-623","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. v1.1 public87tasks,3runs; Anthropic ClaudeCode,OpenAI Codex,Qwen OpenCode."},{"observationId":"launch-qwen-624","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. v1.1 public87tasks,3runs; Anthropic ClaudeCode,OpenAI Codex,Qwen OpenCode."},{"observationId":"launch-qwen-625","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. v1.1 public87tasks,3runs; Anthropic ClaudeCode,OpenAI Codex,Qwen OpenCode."},{"observationId":"launch-qwen-626","modelId":"qwen3.7-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. v1.1 public87tasks,3runs; Anthropic ClaudeCode,OpenAI Codex,Qwen OpenCode."},{"observationId":"launch-qwen-627","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. v1.1 public87tasks,3runs; Anthropic ClaudeCode,OpenAI Codex,Qwen OpenCode."},{"observationId":"launch-qwen-628","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Pass"},{"observationId":"launch-qwen-629","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Score"},{"observationId":"launch-qwen-630","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Pass"},{"observationId":"launch-qwen-631","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Score"},{"observationId":"launch-qwen-632","modelId":"qwen3.7-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Pass"},{"observationId":"launch-qwen-633","modelId":"qwen3.7-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Score"},{"observationId":"launch-qwen-634","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Pass"},{"observationId":"launch-qwen-635","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Score"},{"observationId":"launch-qwen-636","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-637","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-638","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-639","modelId":"qwen3.7-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-640","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-641","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-642","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-643","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-644","modelId":"qwen3.7-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-645","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-646","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Item-F1 over4runs; Qwen-Agent for Qwen,ClaudeCode for comparators."},{"observationId":"launch-qwen-647","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Item-F1 over4runs; Qwen-Agent for Qwen,ClaudeCode for comparators."},{"observationId":"launch-qwen-648","modelId":"qwen3.7-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Item-F1 over4runs; Qwen-Agent for Qwen,ClaudeCode for comparators."},{"observationId":"launch-qwen-649","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card. Item-F1 over4runs; Qwen-Agent for Qwen,ClaudeCode for comparators."},{"observationId":"launch-qwen-650","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-651","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-652","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-653","modelId":"qwen3.7-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-654","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-655","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-656","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-657","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-658","modelId":"qwen3.7-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-659","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-660","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-661","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-662","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-663","modelId":"qwen3.7-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-664","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-665","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-666","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-667","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-668","modelId":"qwen3.7-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-669","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-670","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-671","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-672","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-673","modelId":"qwen3.7-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-674","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-675","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-676","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-677","modelId":"qwen3.7-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-678","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-679","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-680","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-681","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-682","modelId":"qwen3.7-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-683","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-684","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-685","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-686","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-687","modelId":"qwen3.7-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-688","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-689","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-690","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-691","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-692","modelId":"qwen3.7-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-693","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-694","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-695","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-696","modelId":"qwen3.7-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-697","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-698","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-699","modelId":"gpt-5.6-sol","label":"Max","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-700","modelId":"qwen3.7-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-qwen-701","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/README.md","configuration":"Source benchmark-specific methodology; Qwen benchmark effort unspecified, GPT-5.6 Sol max per table header. Max in Qwen model names is not a reported effort setting. Comparator settings and source citations as documented in the model card."},{"observationId":"launch-xai-1318","modelId":"grok-4.6","label":"Not specified","sourceUrl":"https://x.ai/news/grok-4-6","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI."},{"observationId":"launch-xai-1319","modelId":"grok-4.5","label":"Not specified","sourceUrl":"https://x.ai/news/grok-4-6","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI."},{"observationId":"launch-xai-1320","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://x.ai/news/grok-4-6","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI."},{"observationId":"launch-xai-1321","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://x.ai/news/grok-4-6","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI."},{"observationId":"launch-xai-1322","modelId":"grok-4.6","label":"Not specified","sourceUrl":"https://x.ai/news/grok-4-6","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI."},{"observationId":"launch-xai-1323","modelId":"grok-4.5","label":"Not specified","sourceUrl":"https://x.ai/news/grok-4-6","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI."},{"observationId":"launch-xai-1324","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://x.ai/news/grok-4-6","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI."},{"observationId":"launch-xai-1325","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://x.ai/news/grok-4-6","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI."},{"observationId":"launch-xai-1326","modelId":"grok-4.6","label":"Not specified","sourceUrl":"https://x.ai/news/grok-4-6","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI."},{"observationId":"launch-xai-1327","modelId":"grok-4.5","label":"Not specified","sourceUrl":"https://x.ai/news/grok-4-6","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI."},{"observationId":"launch-xai-1328","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://x.ai/news/grok-4-6","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI."},{"observationId":"launch-xai-1329","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://x.ai/news/grok-4-6","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI."},{"observationId":"launch-xai-1330","modelId":"grok-4.6","label":"Not specified","sourceUrl":"https://x.ai/news/grok-4-6","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI."},{"observationId":"launch-xai-1331","modelId":"grok-4.5","label":"Not specified","sourceUrl":"https://x.ai/news/grok-4-6","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI."},{"observationId":"launch-xai-1332","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://x.ai/news/grok-4-6","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI."},{"observationId":"launch-xai-1333","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://x.ai/news/grok-4-6","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI."},{"observationId":"launch-xai-1334","modelId":"grok-4.6","label":"Not specified","sourceUrl":"https://x.ai/news/grok-4-6","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI."},{"observationId":"launch-xai-1335","modelId":"grok-4.5","label":"Not specified","sourceUrl":"https://x.ai/news/grok-4-6","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI."},{"observationId":"launch-xai-1336","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://x.ai/news/grok-4-6","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI."},{"observationId":"launch-xai-1337","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://x.ai/news/grok-4-6","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI."},{"observationId":"launch-xai-1338","modelId":"grok-4.6","label":"Not specified","sourceUrl":"https://x.ai/news/grok-4-6","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI."},{"observationId":"launch-xai-1339","modelId":"grok-4.5","label":"Not specified","sourceUrl":"https://x.ai/news/grok-4-6","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI."},{"observationId":"launch-xai-1340","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://x.ai/news/grok-4-6","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI."},{"observationId":"launch-xai-1341","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://x.ai/news/grok-4-6","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI."},{"observationId":"launch-xai-1342","modelId":"grok-4.6","label":"Not specified","sourceUrl":"https://x.ai/news/grok-4-6","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI."},{"observationId":"launch-xai-1343","modelId":"grok-4.5","label":"Not specified","sourceUrl":"https://x.ai/news/grok-4-6","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI."},{"observationId":"launch-xai-1344","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://x.ai/news/grok-4-6","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI."},{"observationId":"launch-xai-1345","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://x.ai/news/grok-4-6","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI."},{"observationId":"launch-xai-1346","modelId":"grok-4.6","label":"Not specified","sourceUrl":"https://x.ai/news/grok-4-6","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI."},{"observationId":"launch-xai-1347","modelId":"grok-4.5","label":"Not specified","sourceUrl":"https://x.ai/news/grok-4-6","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI."},{"observationId":"launch-xai-1348","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://x.ai/news/grok-4-6","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI."},{"observationId":"launch-xai-1349","modelId":"grok-4.6","label":"Not specified","sourceUrl":"https://x.ai/news/grok-4-6","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI."},{"observationId":"launch-xai-1350","modelId":"grok-4.5","label":"Not specified","sourceUrl":"https://x.ai/news/grok-4-6","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI."},{"observationId":"launch-xai-1351","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://x.ai/news/grok-4-6","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI."},{"observationId":"launch-xai-1352","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://x.ai/news/grok-4-6","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI."},{"observationId":"launch-xai-1353","modelId":"grok-4.6","label":"Not specified","sourceUrl":"https://x.ai/news/grok-4-6","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI."},{"observationId":"launch-xai-1354","modelId":"grok-4.5","label":"Not specified","sourceUrl":"https://x.ai/news/grok-4-6","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI."},{"observationId":"launch-xai-1355","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://x.ai/news/grok-4-6","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI."},{"observationId":"launch-xai-1356","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://x.ai/news/grok-4-6","configuration":"Grok High; GPT-5.6 Sol Max; Fable 5 Max. Competitor figures from developer cards/public leaderboards as selected by xAI."},{"observationId":"launch-zai-1198","modelId":"glm-5.3","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1199","modelId":"glm-5.2","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1200","modelId":"kimi-k3","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1201","modelId":"deepseek-v4-pro","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1202","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1203","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1204","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1205","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1206","modelId":"glm-5.3","label":"Max","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1207","modelId":"glm-5.2","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1208","modelId":"kimi-k3","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1209","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1210","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1211","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1212","modelId":"glm-5.3","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1213","modelId":"glm-5.2","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1214","modelId":"kimi-k3","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1215","modelId":"deepseek-v4-pro","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1216","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1217","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1218","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1219","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1220","modelId":"glm-5.3","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1221","modelId":"glm-5.2","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1222","modelId":"kimi-k3","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1223","modelId":"deepseek-v4-pro","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1224","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1225","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1226","modelId":"glm-5.3","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1227","modelId":"glm-5.2","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1228","modelId":"kimi-k3","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1229","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1230","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1231","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1232","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1233","modelId":"glm-5.3","label":"Max","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1234","modelId":"glm-5.2","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1235","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1236","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1237","modelId":"glm-5.3","label":"Max","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1238","modelId":"glm-5.2","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1239","modelId":"kimi-k3","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1240","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1241","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1242","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1243","modelId":"glm-5.3","label":"Max","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1244","modelId":"glm-5.2","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1245","modelId":"kimi-k3","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1246","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1247","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1248","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1249","modelId":"glm-5.3","label":"Max","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1250","modelId":"glm-5.2","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1251","modelId":"kimi-k3","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1252","modelId":"deepseek-v4-pro","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1253","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1254","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1255","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1256","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1257","modelId":"glm-5.3","label":"Max","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details. 2h Single-run869tasks; API time normalized by model TPS plus overhead; reported number solved; domain whitelist."},{"observationId":"launch-zai-1258","modelId":"glm-5.3","label":"Max","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details. 6h Single-run869tasks; API time normalized by model TPS plus overhead; reported number solved; domain whitelist."},{"observationId":"launch-zai-1259","modelId":"glm-5.2","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details. 2h Single-run869tasks; API time normalized by model TPS plus overhead; reported number solved; domain whitelist."},{"observationId":"launch-zai-1260","modelId":"glm-5.2","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details. 6h Single-run869tasks; API time normalized by model TPS plus overhead; reported number solved; domain whitelist."},{"observationId":"launch-zai-1261","modelId":"kimi-k3","label":"Max","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details. 2h Single-run869tasks; API time normalized by model TPS plus overhead; reported number solved; domain whitelist."},{"observationId":"launch-zai-1262","modelId":"kimi-k3","label":"Max","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details. 6h Single-run869tasks; API time normalized by model TPS plus overhead; reported number solved; domain whitelist."},{"observationId":"launch-zai-1263","modelId":"qwen-3.8-max","label":"Max","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details. 2h Single-run869tasks; API time normalized by model TPS plus overhead; reported number solved; domain whitelist."},{"observationId":"launch-zai-1264","modelId":"qwen-3.8-max","label":"Max","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details. 6h Single-run869tasks; API time normalized by model TPS plus overhead; reported number solved; domain whitelist."},{"observationId":"launch-zai-1265","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details. 2h Single-run869tasks; API time normalized by model TPS plus overhead; reported number solved; domain whitelist."},{"observationId":"launch-zai-1266","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details. 6h Single-run869tasks; API time normalized by model TPS plus overhead; reported number solved; domain whitelist."},{"observationId":"launch-zai-1267","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details. 2h Single-run869tasks; API time normalized by model TPS plus overhead; reported number solved; domain whitelist."},{"observationId":"launch-zai-1268","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details. 6h Single-run869tasks; API time normalized by model TPS plus overhead; reported number solved; domain whitelist."},{"observationId":"launch-zai-1269","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details. 2h Single-run869tasks; API time normalized by model TPS plus overhead; reported number solved; domain whitelist."},{"observationId":"launch-zai-1270","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details. 6h Single-run869tasks; API time normalized by model TPS plus overhead; reported number solved; domain whitelist."},{"observationId":"launch-zai-1271","modelId":"glm-5.3","label":"Max","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1272","modelId":"glm-5.2","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1273","modelId":"kimi-k3","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1274","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1275","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1276","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1277","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1278","modelId":"glm-5.3","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1279","modelId":"glm-5.2","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1280","modelId":"kimi-k3","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1281","modelId":"deepseek-v4-pro","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1282","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1283","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1284","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1285","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1286","modelId":"glm-5.3","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1287","modelId":"glm-5.2","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1288","modelId":"kimi-k3","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1289","modelId":"deepseek-v4-pro","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1290","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1291","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1292","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1293","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1294","modelId":"glm-5.3","label":"Max","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1295","modelId":"glm-5.2","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1296","modelId":"kimi-k3","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1297","modelId":"deepseek-v4-pro","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1298","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1299","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1300","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1301","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1302","modelId":"glm-5.3","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1303","modelId":"glm-5.2","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1304","modelId":"kimi-k3","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1305","modelId":"deepseek-v4-pro","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1306","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1307","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1308","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1309","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1310","modelId":"glm-5.3","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1311","modelId":"glm-5.2","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1312","modelId":"kimi-k3","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1313","modelId":"deepseek-v4-pro","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1314","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1315","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1316","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"launch-zai-1317","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/README.md","configuration":"GLM-5.3 release table; benchmark-specific harness and reasoning configurations in Evaluation Details."},{"observationId":"reported-20260901-claude-fable-5-1-deepswe-v1.1-deepswe-v1.1-anthropic-reported","modelId":"claude-fable-5-1","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":null},{"observationId":"reported-20260901-claude-fable-5-1-hle-hle-no-tools","modelId":"claude-fable-5-1","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":null},{"observationId":"reported-20260901-claude-fable-5-1-hle-hle-with-tools","modelId":"claude-fable-5-1","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":null},{"observationId":"reported-20260901-claude-fable-5-1-swe-bench-pro-swe-bench-pro-anthropic-reported","modelId":"claude-fable-5-1","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":null},{"observationId":"reported-20260901-claude-fable-5-hle-hle-no-tools","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":null},{"observationId":"reported-20260901-claude-fable-5-hle-hle-with-tools","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":null},{"observationId":"reported-20260901-claude-fable-5-swe-bench-pro-swe-bench-pro-anthropic-reported","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":null},{"observationId":"reported-20260901-claude-opus-5-hle-hle-no-tools","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":null},{"observationId":"reported-20260901-claude-opus-5-hle-hle-with-tools","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card","configuration":null},{"observationId":"s-arc2-claude-fable-5-1-efebf61be277","modelId":"claude-fable-5-1","label":"Not specified","sourceUrl":"https://arcprize.org/leaderboard","configuration":null},{"observationId":"s-arc2-claude-fable-5-efebf61be277","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://arcprize.org/leaderboard","configuration":null},{"observationId":"s-arc2-claude-haiku-4-5-20251001-efebf61be277","modelId":"claude-haiku-4-5-20251001","label":"Not specified","sourceUrl":"https://arcprize.org/leaderboard","configuration":null},{"observationId":"s-arc2-claude-opus-4-8-efebf61be277","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://arcprize.org/leaderboard","configuration":null},{"observationId":"s-arc2-claude-opus-5-efebf61be277","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://arcprize.org/leaderboard","configuration":null},{"observationId":"s-arc2-claude-sonnet-4-5-20250929-efebf61be277","modelId":"claude-sonnet-4-5-20250929","label":"Not specified","sourceUrl":"https://arcprize.org/leaderboard","configuration":null},{"observationId":"s-arc2-claude-sonnet-4-6-efebf61be277","modelId":"claude-sonnet-4-6","label":"Not specified","sourceUrl":"https://arcprize.org/leaderboard","configuration":null},{"observationId":"s-arc2-deepseek-v3.2-efebf61be277","modelId":"deepseek-v3.2","label":"Not specified","sourceUrl":"https://arcprize.org/leaderboard","configuration":null},{"observationId":"s-arc2-deepseek-v4-pro-efebf61be277","modelId":"deepseek-v4-pro","label":"Not specified","sourceUrl":"https://arcprize.org/leaderboard","configuration":null},{"observationId":"s-arc2-gemini-2.5-flash-efebf61be277","modelId":"gemini-2.5-flash","label":"Not specified","sourceUrl":"https://arcprize.org/leaderboard","configuration":null},{"observationId":"s-arc2-gemini-3-flash-preview-efebf61be277","modelId":"gemini-3-flash-preview","label":"Not specified","sourceUrl":"https://arcprize.org/leaderboard","configuration":null},{"observationId":"s-arc2-gemini-3.1-pro-efebf61be277","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://arcprize.org/leaderboard","configuration":null},{"observationId":"s-arc2-gemini-3.5-flash-efebf61be277","modelId":"gemini-3.5-flash","label":"Not specified","sourceUrl":"https://arcprize.org/leaderboard","configuration":null},{"observationId":"s-arc2-gemini-3.5-flash-lite-efebf61be277","modelId":"gemini-3.5-flash-lite","label":"Not specified","sourceUrl":"https://arcprize.org/leaderboard","configuration":null},{"observationId":"s-arc2-gemini-3.6-flash-efebf61be277","modelId":"gemini-3.6-flash","label":"Not specified","sourceUrl":"https://arcprize.org/leaderboard","configuration":null},{"observationId":"s-arc2-gemini-3.7-flash-efebf61be277","modelId":"gemini-3.7-flash","label":"Not specified","sourceUrl":"https://arcprize.org/leaderboard","configuration":null},{"observationId":"s-arc2-glm-5-efebf61be277","modelId":"glm-5","label":"Not specified","sourceUrl":"https://arcprize.org/leaderboard","configuration":null},{"observationId":"s-arc2-glm-5.2-efebf61be277","modelId":"glm-5.2","label":"Not specified","sourceUrl":"https://arcprize.org/leaderboard","configuration":null},{"observationId":"s-arc2-gpt-4.1-efebf61be277","modelId":"gpt-4.1","label":"Not specified","sourceUrl":"https://arcprize.org/leaderboard","configuration":null},{"observationId":"s-arc2-gpt-4.1-mini-efebf61be277","modelId":"gpt-4.1-mini","label":"Not specified","sourceUrl":"https://arcprize.org/leaderboard","configuration":null},{"observationId":"s-arc2-gpt-4.1-nano-efebf61be277","modelId":"gpt-4.1-nano","label":"Not specified","sourceUrl":"https://arcprize.org/leaderboard","configuration":null},{"observationId":"s-arc2-gpt-4o-efebf61be277","modelId":"gpt-4o","label":"Not specified","sourceUrl":"https://arcprize.org/leaderboard","configuration":null},{"observationId":"s-arc2-gpt-4o-mini-efebf61be277","modelId":"gpt-4o-mini","label":"Not specified","sourceUrl":"https://arcprize.org/leaderboard","configuration":null},{"observationId":"s-arc2-gpt-5-efebf61be277","modelId":"gpt-5","label":"Not specified","sourceUrl":"https://arcprize.org/leaderboard","configuration":null},{"observationId":"s-arc2-gpt-5-mini-efebf61be277","modelId":"gpt-5-mini","label":"Not specified","sourceUrl":"https://arcprize.org/leaderboard","configuration":null},{"observationId":"s-arc2-gpt-5-nano-efebf61be277","modelId":"gpt-5-nano","label":"Not specified","sourceUrl":"https://arcprize.org/leaderboard","configuration":null},{"observationId":"s-arc2-gpt-5-pro-efebf61be277","modelId":"gpt-5-pro","label":"Not specified","sourceUrl":"https://arcprize.org/leaderboard","configuration":null},{"observationId":"s-arc2-gpt-5.2-efebf61be277","modelId":"gpt-5.2","label":"Not specified","sourceUrl":"https://arcprize.org/leaderboard","configuration":null},{"observationId":"s-arc2-gpt-5.2-pro-efebf61be277","modelId":"gpt-5.2-pro","label":"Not specified","sourceUrl":"https://arcprize.org/leaderboard","configuration":null},{"observationId":"s-arc2-gpt-5.4-efebf61be277","modelId":"gpt-5.4","label":"Not specified","sourceUrl":"https://arcprize.org/leaderboard","configuration":null},{"observationId":"s-arc2-gpt-5.4-mini-efebf61be277","modelId":"gpt-5.4-mini","label":"Not specified","sourceUrl":"https://arcprize.org/leaderboard","configuration":null},{"observationId":"s-arc2-gpt-5.4-nano-efebf61be277","modelId":"gpt-5.4-nano","label":"Not specified","sourceUrl":"https://arcprize.org/leaderboard","configuration":null},{"observationId":"s-arc2-gpt-5.4-pro-efebf61be277","modelId":"gpt-5.4-pro","label":"Not specified","sourceUrl":"https://arcprize.org/leaderboard","configuration":null},{"observationId":"s-arc2-gpt-5.5-efebf61be277","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://arcprize.org/leaderboard","configuration":null},{"observationId":"s-arc2-gpt-5.5-pro-efebf61be277","modelId":"gpt-5.5-pro","label":"Not specified","sourceUrl":"https://arcprize.org/leaderboard","configuration":null},{"observationId":"s-arc2-gpt-5.6-luna-efebf61be277","modelId":"gpt-5.6-luna","label":"Not specified","sourceUrl":"https://arcprize.org/leaderboard","configuration":null},{"observationId":"s-arc2-gpt-5.6-sol-efebf61be277","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://arcprize.org/leaderboard","configuration":null},{"observationId":"s-arc2-gpt-5.6-terra-efebf61be277","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://arcprize.org/leaderboard","configuration":null},{"observationId":"s-arc2-gpt-6-astra-efebf61be277","modelId":"gpt-6-astra","label":"Not specified","sourceUrl":"https://arcprize.org/leaderboard","configuration":null},{"observationId":"s-arc2-grok-4.6-efebf61be277","modelId":"grok-4.6","label":"Not specified","sourceUrl":"https://arcprize.org/leaderboard","configuration":null},{"observationId":"s-arc2-kimi-k3-efebf61be277","modelId":"kimi-k3","label":"Not specified","sourceUrl":"https://arcprize.org/leaderboard","configuration":null},{"observationId":"s-arc2-llama-4-maverick-efebf61be277","modelId":"llama-4-maverick","label":"Not specified","sourceUrl":"https://arcprize.org/leaderboard","configuration":null},{"observationId":"s-arc2-minimax-m2.5-efebf61be277","modelId":"minimax-m2.5","label":"Not specified","sourceUrl":"https://arcprize.org/leaderboard","configuration":null},{"observationId":"s-arc2-o3-efebf61be277","modelId":"o3","label":"Not specified","sourceUrl":"https://arcprize.org/leaderboard","configuration":null},{"observationId":"s-arc2-o3-mini-efebf61be277","modelId":"o3-mini","label":"Not specified","sourceUrl":"https://arcprize.org/leaderboard","configuration":null},{"observationId":"s-arc2-o3-pro-efebf61be277","modelId":"o3-pro","label":"Not specified","sourceUrl":"https://arcprize.org/leaderboard","configuration":null},{"observationId":"s-arc2-o4-mini-efebf61be277","modelId":"o4-mini","label":"Not specified","sourceUrl":"https://arcprize.org/leaderboard","configuration":null},{"observationId":"s-arc2-qwen3-235b-a22b-instruct-2507-efebf61be277","modelId":"qwen3-235b-a22b-instruct-2507","label":"Not specified","sourceUrl":"https://arcprize.org/leaderboard","configuration":null},{"observationId":"s-deepswe-claude-fable-5-eb88d1c756c0","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://deepswe.datacurve.ai/","configuration":null},{"observationId":"s-deepswe-claude-opus-4-8-eb88d1c756c0","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://deepswe.datacurve.ai/","configuration":null},{"observationId":"s-deepswe-claude-opus-5-eb88d1c756c0","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://deepswe.datacurve.ai/","configuration":null},{"observationId":"s-deepswe-claude-sonnet-4-6-eb88d1c756c0","modelId":"claude-sonnet-4-6","label":"Not specified","sourceUrl":"https://deepswe.datacurve.ai/","configuration":null},{"observationId":"s-deepswe-claude-sonnet-5-eb88d1c756c0","modelId":"claude-sonnet-5","label":"Not specified","sourceUrl":"https://deepswe.datacurve.ai/","configuration":null},{"observationId":"s-deepswe-deepseek-v4-flash-eb88d1c756c0","modelId":"deepseek-v4-flash","label":"Not specified","sourceUrl":"https://deepswe.datacurve.ai/","configuration":null},{"observationId":"s-deepswe-deepseek-v4-pro-eb88d1c756c0","modelId":"deepseek-v4-pro","label":"Not specified","sourceUrl":"https://deepswe.datacurve.ai/","configuration":null},{"observationId":"s-deepswe-gemini-3.1-pro-eb88d1c756c0","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://deepswe.datacurve.ai/","configuration":null},{"observationId":"s-deepswe-gemini-3.5-flash-eb88d1c756c0","modelId":"gemini-3.5-flash","label":"Not specified","sourceUrl":"https://deepswe.datacurve.ai/","configuration":null},{"observationId":"s-deepswe-gemini-3.6-flash-eb88d1c756c0","modelId":"gemini-3.6-flash","label":"Not specified","sourceUrl":"https://deepswe.datacurve.ai/","configuration":null},{"observationId":"s-deepswe-gemini-3.7-flash-eb88d1c756c0","modelId":"gemini-3.7-flash","label":"Not specified","sourceUrl":"https://deepswe.datacurve.ai/","configuration":null},{"observationId":"s-deepswe-gemini-3.8-flash-eb88d1c756c0","modelId":"gemini-3.8-flash","label":"Not specified","sourceUrl":"https://deepswe.datacurve.ai/","configuration":null},{"observationId":"s-deepswe-glm-5.2-eb88d1c756c0","modelId":"glm-5.2","label":"Not specified","sourceUrl":"https://deepswe.datacurve.ai/","configuration":null},{"observationId":"s-deepswe-glm-5.3-eb88d1c756c0","modelId":"glm-5.3","label":"Not specified","sourceUrl":"https://deepswe.datacurve.ai/","configuration":null},{"observationId":"s-deepswe-glm-5.3-flash-eb88d1c756c0","modelId":"glm-5.3-flash","label":"Not specified","sourceUrl":"https://deepswe.datacurve.ai/","configuration":null},{"observationId":"s-deepswe-gpt-5.4-eb88d1c756c0","modelId":"gpt-5.4","label":"Not specified","sourceUrl":"https://deepswe.datacurve.ai/","configuration":null},{"observationId":"s-deepswe-gpt-5.5-eb88d1c756c0","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://deepswe.datacurve.ai/","configuration":null},{"observationId":"s-deepswe-gpt-5.6-luna-eb88d1c756c0","modelId":"gpt-5.6-luna","label":"Not specified","sourceUrl":"https://deepswe.datacurve.ai/","configuration":null},{"observationId":"s-deepswe-gpt-5.6-sol-eb88d1c756c0","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://deepswe.datacurve.ai/","configuration":null},{"observationId":"s-deepswe-gpt-5.6-terra-eb88d1c756c0","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://deepswe.datacurve.ai/","configuration":null},{"observationId":"s-deepswe-gpt-6-astra-eb88d1c756c0","modelId":"gpt-6-astra","label":"Not specified","sourceUrl":"https://deepswe.datacurve.ai/","configuration":null},{"observationId":"s-deepswe-grok-4.5-eb88d1c756c0","modelId":"grok-4.5","label":"Not specified","sourceUrl":"https://deepswe.datacurve.ai/","configuration":null},{"observationId":"s-deepswe-grok-4.6-eb88d1c756c0","modelId":"grok-4.6","label":"Not specified","sourceUrl":"https://deepswe.datacurve.ai/","configuration":null},{"observationId":"s-deepswe-kimi-k2.7-code-eb88d1c756c0","modelId":"kimi-k2.7-code","label":"Not specified","sourceUrl":"https://deepswe.datacurve.ai/","configuration":null},{"observationId":"s-deepswe-kimi-k3-eb88d1c756c0","modelId":"kimi-k3","label":"Not specified","sourceUrl":"https://deepswe.datacurve.ai/","configuration":null},{"observationId":"s-deepswe-muse-spark-1.1-eb88d1c756c0","modelId":"muse-spark-1.1","label":"Not specified","sourceUrl":"https://deepswe.datacurve.ai/","configuration":null},{"observationId":"s-deepswe-muse-spark-1.2-eb88d1c756c0","modelId":"muse-spark-1.2","label":"Not specified","sourceUrl":"https://deepswe.datacurve.ai/","configuration":null},{"observationId":"s-deepswe-qwen-3.8-max-eb88d1c756c0","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://deepswe.datacurve.ai/","configuration":null},{"observationId":"s-ds-aa","modelId":"deepseek-v4-pro","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/models/comparisons/deepseek-v4-pro-vs-o3","configuration":null},{"observationId":"s-ds-browse","modelId":"deepseek-v4-pro","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813","configuration":null},{"observationId":"s-ds-hle","modelId":"deepseek-v4-pro","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813","configuration":null},{"observationId":"s-ds-lcb","modelId":"deepseek-v4-pro","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813","configuration":null},{"observationId":"s-ds-swepro","modelId":"deepseek-v4-pro","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813","configuration":null},{"observationId":"s-ds-swev","modelId":"deepseek-v4-pro","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813","configuration":null},{"observationId":"s-ds-tb21","modelId":"deepseek-v4-pro","label":"Not specified","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813","configuration":null},{"observationId":"s-gdpval-agnes-2.5-pro-alpha-58c478670eac","modelId":"agnes-2.5-pro-alpha","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","configuration":null},{"observationId":"s-gdpval-claude-opus-5-58c478670eac","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","configuration":null},{"observationId":"s-gdpval-deepseek-v4-pro-58c478670eac","modelId":"deepseek-v4-pro","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","configuration":null},{"observationId":"s-gdpval-devstral-2512-58c478670eac","modelId":"devstral-2512","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","configuration":null},{"observationId":"s-gdpval-diffusiongemma-26b-a4b-58c478670eac","modelId":"diffusiongemma-26b-a4b","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","configuration":null},{"observationId":"s-gdpval-exaone-4.5-33b-58c478670eac","modelId":"exaone-4.5-33b","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","configuration":null},{"observationId":"s-gdpval-gemini-2.5-pro-58c478670eac","modelId":"gemini-2.5-pro","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","configuration":null},{"observationId":"s-gdpval-gemini-3.1-flash-lite-58c478670eac","modelId":"gemini-3.1-flash-lite","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","configuration":null},{"observationId":"s-gdpval-gemini-3.1-pro-58c478670eac","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","configuration":null},{"observationId":"s-gdpval-gemini-3.5-flash-lite-58c478670eac","modelId":"gemini-3.5-flash-lite","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","configuration":null},{"observationId":"s-gdpval-glm-5.3-flash-58c478670eac","modelId":"glm-5.3-flash","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","configuration":null},{"observationId":"s-gdpval-gpt-4-58c478670eac","modelId":"gpt-4","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","configuration":null},{"observationId":"s-gdpval-gpt-4.1-mini-58c478670eac","modelId":"gpt-4.1-mini","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","configuration":null},{"observationId":"s-gdpval-gpt-4o-mini-58c478670eac","modelId":"gpt-4o-mini","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","configuration":null},{"observationId":"s-gdpval-gpt-5.6-sol-58c478670eac","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","configuration":null},{"observationId":"s-gdpval-grok-4.6-58c478670eac","modelId":"grok-4.6","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","configuration":null},{"observationId":"s-gdpval-hy3-58c478670eac","modelId":"hy3","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","configuration":null},{"observationId":"s-gdpval-kimi-k3-58c478670eac","modelId":"kimi-k3","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","configuration":null},{"observationId":"s-gdpval-labs-devstral-small-2512-58c478670eac","modelId":"labs-devstral-small-2512","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","configuration":null},{"observationId":"s-gdpval-llama-4-maverick-58c478670eac","modelId":"llama-4-maverick","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","configuration":null},{"observationId":"s-gdpval-magistral-medium-2509-58c478670eac","modelId":"magistral-medium-2509","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","configuration":null},{"observationId":"s-gdpval-magistral-small-2509-58c478670eac","modelId":"magistral-small-2509","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","configuration":null},{"observationId":"s-gdpval-mimo-v2.5-58c478670eac","modelId":"mimo-v2.5","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","configuration":null},{"observationId":"s-gdpval-mimo-v2.5-pro-58c478670eac","modelId":"mimo-v2.5-pro","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","configuration":null},{"observationId":"s-gdpval-minicpm5-2b-58c478670eac","modelId":"minicpm5-2b","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","configuration":null},{"observationId":"s-gdpval-minimax-m2.7-58c478670eac","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","configuration":null},{"observationId":"s-gdpval-minimax-m3-58c478670eac","modelId":"minimax-m3","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","configuration":null},{"observationId":"s-gdpval-ministral-3-14b-58c478670eac","modelId":"ministral-3-14b","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","configuration":null},{"observationId":"s-gdpval-ministral-3-3b-58c478670eac","modelId":"ministral-3-3b","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","configuration":null},{"observationId":"s-gdpval-ministral-3-8b-58c478670eac","modelId":"ministral-3-8b","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","configuration":null},{"observationId":"s-gdpval-mistral-large-3-58c478670eac","modelId":"mistral-large-3","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","configuration":null},{"observationId":"s-gdpval-mistral-medium-2508-58c478670eac","modelId":"mistral-medium-2508","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","configuration":null},{"observationId":"s-gdpval-mistral-medium-3.5-58c478670eac","modelId":"mistral-medium-3.5","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","configuration":null},{"observationId":"s-gdpval-mistral-small-2506-58c478670eac","modelId":"mistral-small-2506","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","configuration":null},{"observationId":"s-gdpval-nemotron-3-ultra-58c478670eac","modelId":"nemotron-3-ultra","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","configuration":null},{"observationId":"s-gdpval-qwen-3.8-max-58c478670eac","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","configuration":null},{"observationId":"s-gdpval-qwen3.8-flash-next-58c478670eac","modelId":"qwen3.8-flash-next","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","configuration":null},{"observationId":"s-gdpval-solar-open2-250b-58c478670eac","modelId":"solar-open2-250b","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","configuration":null},{"observationId":"s-gdpval-solar-pro3-260323-58c478670eac","modelId":"solar-pro3-260323","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","configuration":null},{"observationId":"s-gdpval-solar-pro4-260806-58c478670eac","modelId":"solar-pro4-260806","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","configuration":null},{"observationId":"s-gdpval-step-3.7-flash-58c478670eac","modelId":"step-3.7-flash","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","configuration":null},{"observationId":"s-gem31-aa","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/articles/artificial-analysis-intelligence-index-v4-1","configuration":null},{"observationId":"s-gem31-browse","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://deepmind.google/models/model-cards/gemini-3-1-pro/","configuration":null},{"observationId":"s-gem31-swev","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://deepmind.google/models/model-cards/gemini-3-1-pro/","configuration":null},{"observationId":"s-gem31-tau2","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://deepmind.google/models/model-cards/gemini-3-1-pro/","configuration":null},{"observationId":"s-gpqa-claude-fable-5-1-41e862296ea6","modelId":"claude-fable-5-1","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gpqa-diamond","configuration":null},{"observationId":"s-gpqa-claude-fable-5-41e862296ea6","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gpqa-diamond","configuration":null},{"observationId":"s-gpqa-claude-opus-5-41e862296ea6","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gpqa-diamond","configuration":null},{"observationId":"s-gpqa-deepseek-v4-pro-41e862296ea6","modelId":"deepseek-v4-pro","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gpqa-diamond","configuration":null},{"observationId":"s-gpqa-gemini-3.1-pro-41e862296ea6","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gpqa-diamond","configuration":null},{"observationId":"s-gpqa-gemini-3.5-flash-lite-41e862296ea6","modelId":"gemini-3.5-flash-lite","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gpqa-diamond","configuration":null},{"observationId":"s-gpqa-gemini-3.7-flash-41e862296ea6","modelId":"gemini-3.7-flash","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gpqa-diamond","configuration":null},{"observationId":"s-gpqa-gemini-3.8-flash-41e862296ea6","modelId":"gemini-3.8-flash","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gpqa-diamond","configuration":null},{"observationId":"s-gpqa-glm-5.3-41e862296ea6","modelId":"glm-5.3","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gpqa-diamond","configuration":null},{"observationId":"s-gpqa-glm-5.3-flash-41e862296ea6","modelId":"glm-5.3-flash","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gpqa-diamond","configuration":null},{"observationId":"s-gpqa-gpt-5.6-luna-41e862296ea6","modelId":"gpt-5.6-luna","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gpqa-diamond","configuration":null},{"observationId":"s-gpqa-gpt-5.6-sol-41e862296ea6","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gpqa-diamond","configuration":null},{"observationId":"s-gpqa-gpt-5.6-terra-41e862296ea6","modelId":"gpt-5.6-terra","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gpqa-diamond","configuration":null},{"observationId":"s-gpqa-gpt-6-astra-41e862296ea6","modelId":"gpt-6-astra","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gpqa-diamond","configuration":null},{"observationId":"s-gpqa-gpt-oss-120b-41e862296ea6","modelId":"gpt-oss-120b","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gpqa-diamond","configuration":null},{"observationId":"s-gpqa-grok-4.6-41e862296ea6","modelId":"grok-4.6","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gpqa-diamond","configuration":null},{"observationId":"s-gpqa-kimi-k3-41e862296ea6","modelId":"kimi-k3","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gpqa-diamond","configuration":null},{"observationId":"s-gpqa-minimax-m3-41e862296ea6","modelId":"minimax-m3","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gpqa-diamond","configuration":null},{"observationId":"s-gpqa-mistral-medium-3.5-41e862296ea6","modelId":"mistral-medium-3.5","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gpqa-diamond","configuration":null},{"observationId":"s-gpqa-muse-spark-1.3-41e862296ea6","modelId":"muse-spark-1.3","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gpqa-diamond","configuration":null},{"observationId":"s-gpqa-nemotron-3-ultra-41e862296ea6","modelId":"nemotron-3-ultra","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gpqa-diamond","configuration":null},{"observationId":"s-gpqa-qwen3.8-2.4t-a95b-41e862296ea6","modelId":"qwen3.8-2.4t-a95b","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gpqa-diamond","configuration":null},{"observationId":"s-gpqa-qwen3.8-27b-41e862296ea6","modelId":"qwen3.8-27b","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/evaluations/gpqa-diamond","configuration":null},{"observationId":"s-gpt56-aa","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/articles/grok-4-6-benchmarks-and-analysis","configuration":null},{"observationId":"s-gpt56-browse","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":null},{"observationId":"s-gpt56-swepro","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":null},{"observationId":"s-gpt56-tb21","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":null},{"observationId":"s-gpt56-tb21-ultra","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://openai.com/index/gpt-5-6/","configuration":null},{"observationId":"s-grok46-aa","modelId":"grok-4.6","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/articles/grok-4-6-benchmarks-and-analysis","configuration":null},{"observationId":"s-grok46-tb21","modelId":"grok-4.6","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/articles/grok-4-6-benchmarks-and-analysis","configuration":null},{"observationId":"s-hle-claude-opus-4-5-20251101-f549f30a8404","modelId":"claude-opus-4-5-20251101","label":"Not specified","sourceUrl":"https://scale.com/leaderboard/humanitys_last_exam","configuration":null},{"observationId":"s-hle-claude-opus-4-7-f549f30a8404","modelId":"claude-opus-4-7","label":"Not specified","sourceUrl":"https://scale.com/leaderboard/humanitys_last_exam","configuration":null},{"observationId":"s-hle-claude-sonnet-4-5-20250929-f549f30a8404","modelId":"claude-sonnet-4-5-20250929","label":"Not specified","sourceUrl":"https://scale.com/leaderboard/humanitys_last_exam","configuration":null},{"observationId":"s-hle-gemini-3.1-pro-f549f30a8404","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://scale.com/leaderboard/humanitys_last_exam","configuration":null},{"observationId":"s-hle-gemini-3.8-flash-f549f30a8404","modelId":"gemini-3.8-flash","label":"Not specified","sourceUrl":"https://scale.com/leaderboard/humanitys_last_exam","configuration":null},{"observationId":"s-hle-gpt-4.1-f549f30a8404","modelId":"gpt-4.1","label":"Not specified","sourceUrl":"https://scale.com/leaderboard/humanitys_last_exam","configuration":null},{"observationId":"s-hle-llama-4-maverick-f549f30a8404","modelId":"llama-4-maverick","label":"Not specified","sourceUrl":"https://scale.com/leaderboard/humanitys_last_exam","configuration":null},{"observationId":"s-hle-mistral-medium-2505-f549f30a8404","modelId":"mistral-medium-2505","label":"Not specified","sourceUrl":"https://scale.com/leaderboard/humanitys_last_exam","configuration":null},{"observationId":"s-kimi-aa","modelId":"kimi-k3","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/articles/grok-4-6-benchmarks-and-analysis","configuration":null},{"observationId":"s-kimi-browse","modelId":"kimi-k3","label":"Not specified","sourceUrl":"https://www.kimi.ai/blog/kimi-k3","configuration":null},{"observationId":"s-kimi-hle","modelId":"kimi-k3","label":"Not specified","sourceUrl":"https://www.kimi.ai/blog/kimi-k3","configuration":null},{"observationId":"s-kimi-tb21","modelId":"kimi-k3","label":"Not specified","sourceUrl":"https://www.kimi.ai/blog/kimi-k3","configuration":null},{"observationId":"s-lcb-deepseek-r1-0528-ad3f286332a4","modelId":"deepseek-r1-0528","label":"Not specified","sourceUrl":"https://livecodebench.github.io/leaderboard.html","configuration":null},{"observationId":"s-lcb-exaone-4.0-32b-ad3f286332a4","modelId":"exaone-4.0-32b","label":"Not specified","sourceUrl":"https://livecodebench.github.io/leaderboard.html","configuration":null},{"observationId":"s-lcb-gpt-4-turbo-ad3f286332a4","modelId":"gpt-4-turbo","label":"Not specified","sourceUrl":"https://livecodebench.github.io/leaderboard.html","configuration":null},{"observationId":"s-lcb-gpt-4o-ad3f286332a4","modelId":"gpt-4o","label":"Not specified","sourceUrl":"https://livecodebench.github.io/leaderboard.html","configuration":null},{"observationId":"s-lcb-gpt-4o-mini-ad3f286332a4","modelId":"gpt-4o-mini","label":"Not specified","sourceUrl":"https://livecodebench.github.io/leaderboard.html","configuration":null},{"observationId":"s-lcb-o3-ad3f286332a4","modelId":"o3","label":"Not specified","sourceUrl":"https://livecodebench.github.io/leaderboard.html","configuration":null},{"observationId":"s-lcb-o3-mini-ad3f286332a4","modelId":"o3-mini","label":"Not specified","sourceUrl":"https://livecodebench.github.io/leaderboard.html","configuration":null},{"observationId":"s-lcb-o4-mini-ad3f286332a4","modelId":"o4-mini","label":"Not specified","sourceUrl":"https://livecodebench.github.io/leaderboard.html","configuration":null},{"observationId":"s-lcb-qwen3-235b-a22b-ad3f286332a4","modelId":"qwen3-235b-a22b","label":"Not specified","sourceUrl":"https://livecodebench.github.io/leaderboard.html","configuration":null},{"observationId":"s-llama4-gpqa","modelId":"llama-4-maverick","label":"Not specified","sourceUrl":"https://ai.meta.com/blog/llama-4-multimodal-intelligence/","configuration":null},{"observationId":"s-llama4-lcb","modelId":"llama-4-maverick","label":"Not specified","sourceUrl":"https://ai.meta.com/blog/llama-4-multimodal-intelligence/","configuration":null},{"observationId":"s-lmarena-c4ai-aya-expanse-32b-465fe8c26c46","modelId":"c4ai-aya-expanse-32b","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-c4ai-aya-expanse-8b-465fe8c26c46","modelId":"c4ai-aya-expanse-8b","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-chatglm-6b-465fe8c26c46","modelId":"chatglm-6b","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-chatglm2-6b-465fe8c26c46","modelId":"chatglm2-6b","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-chatglm3-6b-465fe8c26c46","modelId":"chatglm3-6b","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-claude-fable-5-465fe8c26c46","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-claude-haiku-4-5-20251001-465fe8c26c46","modelId":"claude-haiku-4-5-20251001","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-claude-opus-4-5-20251101-465fe8c26c46","modelId":"claude-opus-4-5-20251101","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-claude-opus-4-6-465fe8c26c46","modelId":"claude-opus-4-6","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-claude-opus-4-7-465fe8c26c46","modelId":"claude-opus-4-7","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-claude-opus-4-8-465fe8c26c46","modelId":"claude-opus-4-8","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-claude-opus-5-465fe8c26c46","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-claude-sonnet-4-5-20250929-465fe8c26c46","modelId":"claude-sonnet-4-5-20250929","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-claude-sonnet-4-6-465fe8c26c46","modelId":"claude-sonnet-4-6","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-command-a-03-2025-465fe8c26c46","modelId":"command-a-03-2025","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-command-r-03-2024-465fe8c26c46","modelId":"command-r-03-2024","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-command-r-08-2024-465fe8c26c46","modelId":"command-r-08-2024","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-command-r-plus-04-2024-465fe8c26c46","modelId":"command-r-plus-04-2024","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-command-r-plus-08-2024-465fe8c26c46","modelId":"command-r-plus-08-2024","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-deepseek-r1-0528-465fe8c26c46","modelId":"deepseek-r1-0528","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-deepseek-v3.2-465fe8c26c46","modelId":"deepseek-v3.2","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-deepseek-v4-flash-465fe8c26c46","modelId":"deepseek-v4-flash","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-deepseek-v4-pro-465fe8c26c46","modelId":"deepseek-v4-pro","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-ernie-5.1-465fe8c26c46","modelId":"ernie-5.1","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-gemini-2.5-flash-465fe8c26c46","modelId":"gemini-2.5-flash","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-gemini-2.5-pro-465fe8c26c46","modelId":"gemini-2.5-pro","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-gemini-3.1-pro-465fe8c26c46","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-gemini-3.5-flash-lite-465fe8c26c46","modelId":"gemini-3.5-flash-lite","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-gemma-1.1-2b-it-465fe8c26c46","modelId":"gemma-1.1-2b-it","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-gemma-1.1-7b-it-465fe8c26c46","modelId":"gemma-1.1-7b-it","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-gemma-2-27b-it-465fe8c26c46","modelId":"gemma-2-27b-it","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-gemma-2-2b-it-465fe8c26c46","modelId":"gemma-2-2b-it","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-gemma-2-9b-it-465fe8c26c46","modelId":"gemma-2-9b-it","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-gemma-3-12b-it-465fe8c26c46","modelId":"gemma-3-12b-it","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-gemma-3-27b-it-465fe8c26c46","modelId":"gemma-3-27b-it","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-gemma-3-4b-it-465fe8c26c46","modelId":"gemma-3-4b-it","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-gemma-3n-e4b-it-465fe8c26c46","modelId":"gemma-3n-e4b-it","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-glm-4.5-465fe8c26c46","modelId":"glm-4.5","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-glm-4.5-air-465fe8c26c46","modelId":"glm-4.5-air","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-glm-4.5v-465fe8c26c46","modelId":"glm-4.5v","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-glm-4.6-465fe8c26c46","modelId":"glm-4.6","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-glm-4.6v-465fe8c26c46","modelId":"glm-4.6v","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-glm-4.7-465fe8c26c46","modelId":"glm-4.7","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-glm-4.7-flash-465fe8c26c46","modelId":"glm-4.7-flash","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-glm-5-465fe8c26c46","modelId":"glm-5","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-glm-5.1-465fe8c26c46","modelId":"glm-5.1","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-glm-5.3-flash-465fe8c26c46","modelId":"glm-5.3-flash","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-glm-5v-turbo-465fe8c26c46","modelId":"glm-5v-turbo","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-gpt-3.5-turbo-1106-465fe8c26c46","modelId":"gpt-3.5-turbo-1106","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-gpt-5.1-465fe8c26c46","modelId":"gpt-5.1","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-gpt-5.2-465fe8c26c46","modelId":"gpt-5.2","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-gpt-5.4-465fe8c26c46","modelId":"gpt-5.4","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-gpt-5.5-465fe8c26c46","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-gpt-5.6-sol-465fe8c26c46","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-gpt-oss-120b-465fe8c26c46","modelId":"gpt-oss-120b","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-gpt-oss-20b-465fe8c26c46","modelId":"gpt-oss-20b","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-grok-4.3-465fe8c26c46","modelId":"grok-4.3","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-grok-4.5-465fe8c26c46","modelId":"grok-4.5","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-grok-4.6-465fe8c26c46","modelId":"grok-4.6","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-hy3-465fe8c26c46","modelId":"hy3","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-kimi-k2.6-465fe8c26c46","modelId":"kimi-k2.6","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-kimi-k3-465fe8c26c46","modelId":"kimi-k3","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-llama-3.1-70b-instruct-465fe8c26c46","modelId":"llama-3.1-70b-instruct","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-llama-3.1-8b-instruct-465fe8c26c46","modelId":"llama-3.1-8b-instruct","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-llama-3.2-1b-instruct-465fe8c26c46","modelId":"llama-3.2-1b-instruct","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-llama-3.2-3b-instruct-465fe8c26c46","modelId":"llama-3.2-3b-instruct","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-llama-3.3-70b-instruct-465fe8c26c46","modelId":"llama-3.3-70b-instruct","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-llama-4-maverick-465fe8c26c46","modelId":"llama-4-maverick","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-llama-4-scout-465fe8c26c46","modelId":"llama-4-scout","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-mimo-v2.5-465fe8c26c46","modelId":"mimo-v2.5","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-mimo-v2.5-pro-465fe8c26c46","modelId":"mimo-v2.5-pro","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-minimax-m2-465fe8c26c46","modelId":"minimax-m2","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-minimax-m2.5-465fe8c26c46","modelId":"minimax-m2.5","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-minimax-m2.7-465fe8c26c46","modelId":"minimax-m2.7","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-minimax-m3-465fe8c26c46","modelId":"minimax-m3","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-ministral-8b-2410-465fe8c26c46","modelId":"ministral-8b-2410","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-mistral-large-2407-465fe8c26c46","modelId":"mistral-large-2407","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-mistral-large-2411-465fe8c26c46","modelId":"mistral-large-2411","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-mistral-large-3-465fe8c26c46","modelId":"mistral-large-3","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-mistral-medium-2505-465fe8c26c46","modelId":"mistral-medium-2505","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-mistral-medium-2508-465fe8c26c46","modelId":"mistral-medium-2508","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-mistral-medium-3.5-465fe8c26c46","modelId":"mistral-medium-3.5","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-mistral-small-2506-465fe8c26c46","modelId":"mistral-small-2506","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-muse-spark-1.1-465fe8c26c46","modelId":"muse-spark-1.1","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-nemotron-3-ultra-465fe8c26c46","modelId":"nemotron-3-ultra","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-nvidia-nemotron-3-nano-30b-a3b-bf16-465fe8c26c46","modelId":"nvidia-nemotron-3-nano-30b-a3b-bf16","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-o3-mini-465fe8c26c46","modelId":"o3-mini","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-phi-3-medium-4k-instruct-465fe8c26c46","modelId":"phi-3-medium-4k-instruct","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-phi-3-mini-128k-instruct-465fe8c26c46","modelId":"phi-3-mini-128k-instruct","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-phi-3-mini-4k-instruct-465fe8c26c46","modelId":"phi-3-mini-4k-instruct","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-phi-3-small-8k-instruct-465fe8c26c46","modelId":"phi-3-small-8k-instruct","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-phi-4-465fe8c26c46","modelId":"phi-4","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-qwen-3.8-max-465fe8c26c46","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-qwen3-235b-a22b-465fe8c26c46","modelId":"qwen3-235b-a22b","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-qwen3-235b-a22b-instruct-2507-465fe8c26c46","modelId":"qwen3-235b-a22b-instruct-2507","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-qwen3-235b-a22b-thinking-2507-465fe8c26c46","modelId":"qwen3-235b-a22b-thinking-2507","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-qwen3-30b-a3b-465fe8c26c46","modelId":"qwen3-30b-a3b","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-qwen3-30b-a3b-instruct-2507-465fe8c26c46","modelId":"qwen3-30b-a3b-instruct-2507","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-qwen3-32b-465fe8c26c46","modelId":"qwen3-32b","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-qwen3-coder-480b-a35b-instruct-465fe8c26c46","modelId":"qwen3-coder-480b-a35b-instruct","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-qwen3-next-80b-a3b-instruct-465fe8c26c46","modelId":"qwen3-next-80b-a3b-instruct","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-qwen3-next-80b-a3b-thinking-465fe8c26c46","modelId":"qwen3-next-80b-a3b-thinking","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-qwen3-vl-235b-a22b-instruct-465fe8c26c46","modelId":"qwen3-vl-235b-a22b-instruct","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-qwen3-vl-235b-a22b-thinking-465fe8c26c46","modelId":"qwen3-vl-235b-a22b-thinking","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-qwen3.5-122b-a10b-465fe8c26c46","modelId":"qwen3.5-122b-a10b","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-qwen3.5-27b-465fe8c26c46","modelId":"qwen3.5-27b","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-qwen3.5-35b-a3b-465fe8c26c46","modelId":"qwen3.5-35b-a3b","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-qwen3.5-397b-a17b-465fe8c26c46","modelId":"qwen3.5-397b-a17b","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-qwen3.5-flash-465fe8c26c46","modelId":"qwen3.5-flash","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-qwen3.6-plus-465fe8c26c46","modelId":"qwen3.6-plus","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-qwen3.7-plus-465fe8c26c46","modelId":"qwen3.7-plus","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-qwen3.8-27b-465fe8c26c46","modelId":"qwen3.8-27b","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-solar-10.7b-instruct-v1.0-465fe8c26c46","modelId":"solar-10.7b-instruct-v1.0","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-solar-pro4-260806-465fe8c26c46","modelId":"solar-pro4-260806","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-lmarena-step-3.5-flash-465fe8c26c46","modelId":"step-3.5-flash","label":"Not specified","sourceUrl":"https://lmarena.ai/leaderboard","configuration":null},{"observationId":"s-mistral-gpqa","modelId":"mistral-large-3","label":"Not specified","sourceUrl":"https://vals.ai/models/mistralai_mistral-large-2512","configuration":null},{"observationId":"s-mmlu-pro-c4ai-aya-expanse-8b-df03094a9acf","modelId":"c4ai-aya-expanse-8b","label":"Not specified","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro","configuration":null},{"observationId":"s-mmlu-pro-deepseek-r1-0528-df03094a9acf","modelId":"deepseek-r1-0528","label":"Not specified","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro","configuration":null},{"observationId":"s-mmlu-pro-exaone-3.5-2.4b-instruct-df03094a9acf","modelId":"exaone-3.5-2.4b-instruct","label":"Not specified","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro","configuration":null},{"observationId":"s-mmlu-pro-exaone-3.5-32b-instruct-df03094a9acf","modelId":"exaone-3.5-32b-instruct","label":"Not specified","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro","configuration":null},{"observationId":"s-mmlu-pro-exaone-3.5-7.8b-instruct-df03094a9acf","modelId":"exaone-3.5-7.8b-instruct","label":"Not specified","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro","configuration":null},{"observationId":"s-mmlu-pro-gemini-2.5-pro-df03094a9acf","modelId":"gemini-2.5-pro","label":"Not specified","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro","configuration":null},{"observationId":"s-mmlu-pro-gemini-3.1-pro-df03094a9acf","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro","configuration":null},{"observationId":"s-mmlu-pro-gemma-2-27b-it-df03094a9acf","modelId":"gemma-2-27b-it","label":"Not specified","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro","configuration":null},{"observationId":"s-mmlu-pro-gemma-2-2b-it-df03094a9acf","modelId":"gemma-2-2b-it","label":"Not specified","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro","configuration":null},{"observationId":"s-mmlu-pro-gemma-2-9b-it-df03094a9acf","modelId":"gemma-2-9b-it","label":"Not specified","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro","configuration":null},{"observationId":"s-mmlu-pro-gemma-3-12b-it-df03094a9acf","modelId":"gemma-3-12b-it","label":"Not specified","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro","configuration":null},{"observationId":"s-mmlu-pro-gemma-3-1b-it-df03094a9acf","modelId":"gemma-3-1b-it","label":"Not specified","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro","configuration":null},{"observationId":"s-mmlu-pro-gemma-3-27b-it-df03094a9acf","modelId":"gemma-3-27b-it","label":"Not specified","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro","configuration":null},{"observationId":"s-mmlu-pro-gemma-3-4b-it-df03094a9acf","modelId":"gemma-3-4b-it","label":"Not specified","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro","configuration":null},{"observationId":"s-mmlu-pro-glm-4-9b-chat-df03094a9acf","modelId":"glm-4-9b-chat","label":"Not specified","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro","configuration":null},{"observationId":"s-mmlu-pro-glm-4.5-air-df03094a9acf","modelId":"glm-4.5-air","label":"Not specified","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro","configuration":null},{"observationId":"s-mmlu-pro-glm-4.5-df03094a9acf","modelId":"glm-4.5","label":"Not specified","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro","configuration":null},{"observationId":"s-mmlu-pro-glm-5-df03094a9acf","modelId":"glm-5","label":"Not specified","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro","configuration":null},{"observationId":"s-mmlu-pro-gpt-4-turbo-df03094a9acf","modelId":"gpt-4-turbo","label":"Not specified","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro","configuration":null},{"observationId":"s-mmlu-pro-gpt-4.1-df03094a9acf","modelId":"gpt-4.1","label":"Not specified","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro","configuration":null},{"observationId":"s-mmlu-pro-gpt-4o-mini-df03094a9acf","modelId":"gpt-4o-mini","label":"Not specified","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro","configuration":null},{"observationId":"s-mmlu-pro-gpt-5.1-df03094a9acf","modelId":"gpt-5.1","label":"Not specified","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro","configuration":null},{"observationId":"s-mmlu-pro-gpt-5.2-df03094a9acf","modelId":"gpt-5.2","label":"Not specified","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro","configuration":null},{"observationId":"s-mmlu-pro-gpt-5.4-df03094a9acf","modelId":"gpt-5.4","label":"Not specified","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro","configuration":null},{"observationId":"s-mmlu-pro-llama-3.1-405b-instruct-df03094a9acf","modelId":"llama-3.1-405b-instruct","label":"Not specified","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro","configuration":null},{"observationId":"s-mmlu-pro-llama-3.1-70b-instruct-df03094a9acf","modelId":"llama-3.1-70b-instruct","label":"Not specified","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro","configuration":null},{"observationId":"s-mmlu-pro-llama-3.1-8b-instruct-df03094a9acf","modelId":"llama-3.1-8b-instruct","label":"Not specified","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro","configuration":null},{"observationId":"s-mmlu-pro-llama-3.3-70b-instruct-df03094a9acf","modelId":"llama-3.3-70b-instruct","label":"Not specified","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro","configuration":null},{"observationId":"s-mmlu-pro-llama-4-maverick-df03094a9acf","modelId":"llama-4-maverick","label":"Not specified","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro","configuration":null},{"observationId":"s-mmlu-pro-llama-4-scout-df03094a9acf","modelId":"llama-4-scout","label":"Not specified","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro","configuration":null},{"observationId":"s-mmlu-pro-mimo-7b-rl-df03094a9acf","modelId":"mimo-7b-rl","label":"Not specified","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro","configuration":null},{"observationId":"s-mmlu-pro-minimax-m2-df03094a9acf","modelId":"minimax-m2","label":"Not specified","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro","configuration":null},{"observationId":"s-mmlu-pro-minimax-m2.1-df03094a9acf","modelId":"minimax-m2.1","label":"Not specified","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro","configuration":null},{"observationId":"s-mmlu-pro-minimax-m2.5-df03094a9acf","modelId":"minimax-m2.5","label":"Not specified","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro","configuration":null},{"observationId":"s-mmlu-pro-minimax-text-01-df03094a9acf","modelId":"minimax-text-01","label":"Not specified","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro","configuration":null},{"observationId":"s-mmlu-pro-phi-3.5-mini-instruct-df03094a9acf","modelId":"phi-3.5-mini-instruct","label":"Not specified","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro","configuration":null},{"observationId":"s-mmlu-pro-phi-4-df03094a9acf","modelId":"phi-4","label":"Not specified","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro","configuration":null},{"observationId":"s-mmlu-pro-phi-4-reasoning-df03094a9acf","modelId":"phi-4-reasoning","label":"Not specified","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro","configuration":null},{"observationId":"s-mmlu-pro-phi-4-reasoning-plus-df03094a9acf","modelId":"phi-4-reasoning-plus","label":"Not specified","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro","configuration":null},{"observationId":"s-mmlu-pro-qwen3-235b-a22b-df03094a9acf","modelId":"qwen3-235b-a22b","label":"Not specified","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro","configuration":null},{"observationId":"s-mmlu-pro-qwen3-235b-a22b-instruct-2507-df03094a9acf","modelId":"qwen3-235b-a22b-instruct-2507","label":"Not specified","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro","configuration":null},{"observationId":"s-mmlu-pro-qwen3-235b-a22b-thinking-2507-df03094a9acf","modelId":"qwen3-235b-a22b-thinking-2507","label":"Not specified","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro","configuration":null},{"observationId":"s-mmlu-pro-qwen3-30b-a3b-thinking-2507-df03094a9acf","modelId":"qwen3-30b-a3b-thinking-2507","label":"Not specified","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro","configuration":null},{"observationId":"s-mmlu-pro-qwen3.5-0.8b-df03094a9acf","modelId":"qwen3.5-0.8b","label":"Not specified","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro","configuration":null},{"observationId":"s-mmlu-pro-qwen3.5-122b-a10b-df03094a9acf","modelId":"qwen3.5-122b-a10b","label":"Not specified","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro","configuration":null},{"observationId":"s-mmlu-pro-qwen3.5-27b-df03094a9acf","modelId":"qwen3.5-27b","label":"Not specified","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro","configuration":null},{"observationId":"s-mmlu-pro-qwen3.5-2b-df03094a9acf","modelId":"qwen3.5-2b","label":"Not specified","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro","configuration":null},{"observationId":"s-mmlu-pro-qwen3.5-35b-a3b-df03094a9acf","modelId":"qwen3.5-35b-a3b","label":"Not specified","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro","configuration":null},{"observationId":"s-mmlu-pro-qwen3.5-397b-a17b-df03094a9acf","modelId":"qwen3.5-397b-a17b","label":"Not specified","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro","configuration":null},{"observationId":"s-mmlu-pro-qwen3.5-4b-df03094a9acf","modelId":"qwen3.5-4b","label":"Not specified","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro","configuration":null},{"observationId":"s-mmlu-pro-qwen3.5-9b-df03094a9acf","modelId":"qwen3.5-9b","label":"Not specified","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro","configuration":null},{"observationId":"s-mmlu-pro-seed-oss-36b-instruct-df03094a9acf","modelId":"seed-oss-36b-instruct","label":"Not specified","sourceUrl":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro","configuration":null},{"observationId":"s-nv-aa","modelId":"nemotron-3-ultra","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/models/comparisons/nvidia-nemotron-3-ultra-550b-a55b-vs-kimi-k2-6","configuration":null},{"observationId":"s-nv-browse","modelId":"nemotron-3-ultra","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-NVFP4","configuration":null},{"observationId":"s-nv-hle","modelId":"nemotron-3-ultra","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-NVFP4","configuration":null},{"observationId":"s-nv-lcb","modelId":"nemotron-3-ultra","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-NVFP4","configuration":null},{"observationId":"s-nv-mmlu","modelId":"nemotron-3-ultra","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-NVFP4","configuration":null},{"observationId":"s-nv-swev","modelId":"nemotron-3-ultra","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-NVFP4","configuration":null},{"observationId":"s-nv-tau2","modelId":"nemotron-3-ultra","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-NVFP4","configuration":null},{"observationId":"s-nv-tb21","modelId":"nemotron-3-ultra","label":"Not specified","sourceUrl":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-NVFP4","configuration":null},{"observationId":"s-opus5-aa","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://artificialanalysis.ai/articles/grok-4-6-benchmarks-and-analysis","configuration":null},{"observationId":"s-opus5-browse","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://www.anthropic.com/news/claude-opus-5","configuration":null},{"observationId":"s-opus5-hle","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://www.anthropic.com/news/claude-opus-5","configuration":null},{"observationId":"s-opus5-hle-tools","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://www.anthropic.com/news/claude-opus-5","configuration":null},{"observationId":"s-opus5-swepro","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://www.anthropic.com/news/claude-opus-5","configuration":null},{"observationId":"s-opus5-swev","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://www.anthropic.com/news/claude-opus-5","configuration":null},{"observationId":"s-osworld-claude-fable-5-42c386263956","modelId":"claude-fable-5","label":"Not specified","sourceUrl":"https://os-world.github.io/","configuration":null},{"observationId":"s-osworld-claude-opus-5-42c386263956","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://os-world.github.io/","configuration":null},{"observationId":"s-osworld-claude-sonnet-4-5-20250929-42c386263956","modelId":"claude-sonnet-4-5-20250929","label":"Not specified","sourceUrl":"https://os-world.github.io/","configuration":null},{"observationId":"s-osworld-claude-sonnet-4-6-42c386263956","modelId":"claude-sonnet-4-6","label":"Not specified","sourceUrl":"https://os-world.github.io/","configuration":null},{"observationId":"s-osworld-muse-spark-1.1-42c386263956","modelId":"muse-spark-1.1","label":"Not specified","sourceUrl":"https://os-world.github.io/","configuration":null},{"observationId":"s-osworld-o3-42c386263956","modelId":"o3","label":"Not specified","sourceUrl":"https://os-world.github.io/","configuration":null},{"observationId":"s-qwen-aa","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://x.com/ArtificialAnlys/status/2085270415614828675","configuration":null},{"observationId":"s-qwen-gpqa","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://qwen.ai/blog?id=qwen3-max","configuration":null},{"observationId":"s-qwen-hle","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://qwen.ai/blog?id=qwen3-max","configuration":null},{"observationId":"s-qwen-osworld","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://qwen.ai/blog?id=qwen3-max","configuration":null},{"observationId":"s-qwen-paper","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://qwen.ai/blog?id=qwen3-max","configuration":null},{"observationId":"s-qwen-swepro","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://qwen.ai/blog?id=qwen3-max","configuration":null},{"observationId":"s-qwen-tb21","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://qwen.ai/blog?id=qwen3-max","configuration":null},{"observationId":"s-swepro-claude-opus-4-6-6112d01393e8","modelId":"claude-opus-4-6","label":"Not specified","sourceUrl":"https://scale.com/leaderboard/swe_bench_pro_public","configuration":null},{"observationId":"s-swepro-gemini-3.1-pro-6112d01393e8","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://scale.com/leaderboard/swe_bench_pro_public","configuration":null},{"observationId":"s-swepro-gpt-5.4-6112d01393e8","modelId":"gpt-5.4","label":"Not specified","sourceUrl":"https://scale.com/leaderboard/swe_bench_pro_public","configuration":null},{"observationId":"s-swepro-muse-spark-1.1-6112d01393e8","modelId":"muse-spark-1.1","label":"Not specified","sourceUrl":"https://scale.com/leaderboard/swe_bench_pro_public","configuration":null},{"observationId":"s-swev-claude-haiku-4-5-20251001-077d8c0cfcca","modelId":"claude-haiku-4-5-20251001","label":"Not specified","sourceUrl":"https://www.swebench.com/","configuration":null},{"observationId":"s-swev-claude-opus-4-5-20251101-077d8c0cfcca","modelId":"claude-opus-4-5-20251101","label":"Not specified","sourceUrl":"https://www.swebench.com/","configuration":null},{"observationId":"s-swev-claude-opus-4-6-077d8c0cfcca","modelId":"claude-opus-4-6","label":"Not specified","sourceUrl":"https://www.swebench.com/","configuration":null},{"observationId":"s-swev-claude-sonnet-4-5-20250929-077d8c0cfcca","modelId":"claude-sonnet-4-5-20250929","label":"Not specified","sourceUrl":"https://www.swebench.com/","configuration":null},{"observationId":"s-swev-deepseek-v3.2-077d8c0cfcca","modelId":"deepseek-v3.2","label":"Not specified","sourceUrl":"https://www.swebench.com/","configuration":null},{"observationId":"s-swev-devstral-2512-077d8c0cfcca","modelId":"devstral-2512","label":"Not specified","sourceUrl":"https://www.swebench.com/","configuration":null},{"observationId":"s-swev-gemini-2.5-flash-077d8c0cfcca","modelId":"gemini-2.5-flash","label":"Not specified","sourceUrl":"https://www.swebench.com/","configuration":null},{"observationId":"s-swev-gemini-2.5-pro-077d8c0cfcca","modelId":"gemini-2.5-pro","label":"Not specified","sourceUrl":"https://www.swebench.com/","configuration":null},{"observationId":"s-swev-gemini-3-flash-preview-077d8c0cfcca","modelId":"gemini-3-flash-preview","label":"Not specified","sourceUrl":"https://www.swebench.com/","configuration":null},{"observationId":"s-swev-glm-4.5-077d8c0cfcca","modelId":"glm-4.5","label":"Not specified","sourceUrl":"https://www.swebench.com/","configuration":null},{"observationId":"s-swev-glm-4.6-077d8c0cfcca","modelId":"glm-4.6","label":"Not specified","sourceUrl":"https://www.swebench.com/","configuration":null},{"observationId":"s-swev-glm-5-077d8c0cfcca","modelId":"glm-5","label":"Not specified","sourceUrl":"https://www.swebench.com/","configuration":null},{"observationId":"s-swev-gpt-oss-120b-077d8c0cfcca","modelId":"gpt-oss-120b","label":"Not specified","sourceUrl":"https://www.swebench.com/","configuration":null},{"observationId":"s-swev-llama-4-maverick-077d8c0cfcca","modelId":"llama-4-maverick","label":"Not specified","sourceUrl":"https://www.swebench.com/","configuration":null},{"observationId":"s-swev-minimax-m2-077d8c0cfcca","modelId":"minimax-m2","label":"Not specified","sourceUrl":"https://www.swebench.com/","configuration":null},{"observationId":"s-swev-minimax-m2.5-077d8c0cfcca","modelId":"minimax-m2.5","label":"Not specified","sourceUrl":"https://www.swebench.com/","configuration":null},{"observationId":"s-swev-o3-077d8c0cfcca","modelId":"o3","label":"Not specified","sourceUrl":"https://www.swebench.com/","configuration":null},{"observationId":"s-swev-o4-mini-077d8c0cfcca","modelId":"o4-mini","label":"Not specified","sourceUrl":"https://www.swebench.com/","configuration":null},{"observationId":"s-swev-qwen3-coder-480b-a35b-instruct-077d8c0cfcca","modelId":"qwen3-coder-480b-a35b-instruct","label":"Not specified","sourceUrl":"https://www.swebench.com/","configuration":null},{"observationId":"s-tb21-gemini-3.1-pro-a98710a64150","modelId":"gemini-3.1-pro","label":"Not specified","sourceUrl":"https://www.tbench.ai/?version=2.1","configuration":null},{"observationId":"s-tb21-gpt-5.5-a98710a64150","modelId":"gpt-5.5","label":"Not specified","sourceUrl":"https://www.tbench.ai/?version=2.1","configuration":null},{"observationId":"s-vulcanbench-claude-opus-5-10-opus5-effort-high-2f58434bbe3d","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://vulcanbench.com/benchmarks/10-opus5-effort.html","configuration":null},{"observationId":"s-vulcanbench-claude-opus-5-10-opus5-effort-low-2f58434bbe3d","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://vulcanbench.com/benchmarks/10-opus5-effort.html","configuration":null},{"observationId":"s-vulcanbench-claude-opus-5-10-opus5-effort-medium-2f58434bbe3d","modelId":"claude-opus-5","label":"Not specified","sourceUrl":"https://vulcanbench.com/benchmarks/10-opus5-effort.html","configuration":null},{"observationId":"s-vulcanbench-glm-5.3-18-glm53-zcode-harness-high-2f58434bbe3d","modelId":"glm-5.3","label":"Not specified","sourceUrl":"https://vulcanbench.com/benchmarks/18-glm53-zcode-harness.html","configuration":null},{"observationId":"s-vulcanbench-glm-5.3-18-glm53-zcode-harness-low-2f58434bbe3d","modelId":"glm-5.3","label":"Not specified","sourceUrl":"https://vulcanbench.com/benchmarks/18-glm53-zcode-harness.html","configuration":null},{"observationId":"s-vulcanbench-glm-5.3-18-glm53-zcode-harness-max-2f58434bbe3d","modelId":"glm-5.3","label":"Not specified","sourceUrl":"https://vulcanbench.com/benchmarks/18-glm53-zcode-harness.html","configuration":null},{"observationId":"s-vulcanbench-gpt-5.6-sol-07-grok-fable-sol-high-2f58434bbe3d","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://vulcanbench.com/benchmarks/07-grok-fable-sol.html","configuration":null},{"observationId":"s-vulcanbench-gpt-5.6-sol-07-grok-fable-sol-low-2f58434bbe3d","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://vulcanbench.com/benchmarks/07-grok-fable-sol.html","configuration":null},{"observationId":"s-vulcanbench-gpt-5.6-sol-07-grok-fable-sol-medium-2f58434bbe3d","modelId":"gpt-5.6-sol","label":"Not specified","sourceUrl":"https://vulcanbench.com/benchmarks/07-grok-fable-sol.html","configuration":null},{"observationId":"s-vulcanbench-grok-4.5-07-grok-fable-sol-high-2f58434bbe3d","modelId":"grok-4.5","label":"Not specified","sourceUrl":"https://vulcanbench.com/benchmarks/07-grok-fable-sol.html","configuration":null},{"observationId":"s-vulcanbench-grok-4.5-07-grok-fable-sol-low-2f58434bbe3d","modelId":"grok-4.5","label":"Not specified","sourceUrl":"https://vulcanbench.com/benchmarks/07-grok-fable-sol.html","configuration":null},{"observationId":"s-vulcanbench-grok-4.5-07-grok-fable-sol-medium-2f58434bbe3d","modelId":"grok-4.5","label":"Not specified","sourceUrl":"https://vulcanbench.com/benchmarks/07-grok-fable-sol.html","configuration":null},{"observationId":"s-vulcanbench-grok-4.6-14-grok-46-effort-high-2f58434bbe3d","modelId":"grok-4.6","label":"Not specified","sourceUrl":"https://vulcanbench.com/benchmarks/14-grok-46-effort.html","configuration":null},{"observationId":"s-vulcanbench-grok-4.6-14-grok-46-effort-low-2f58434bbe3d","modelId":"grok-4.6","label":"Not specified","sourceUrl":"https://vulcanbench.com/benchmarks/14-grok-46-effort.html","configuration":null},{"observationId":"s-vulcanbench-grok-4.6-14-grok-46-effort-medium-2f58434bbe3d","modelId":"grok-4.6","label":"Not specified","sourceUrl":"https://vulcanbench.com/benchmarks/14-grok-46-effort.html","configuration":null},{"observationId":"s-vulcanbench-grok-4.6-14-grok-46-effort-xhigh-2f58434bbe3d","modelId":"grok-4.6","label":"Not specified","sourceUrl":"https://vulcanbench.com/benchmarks/14-grok-46-effort.html","configuration":null},{"observationId":"s-vulcanbench-kimi-k3-08-kimi-k3-max-2f58434bbe3d","modelId":"kimi-k3","label":"Not specified","sourceUrl":"https://vulcanbench.com/benchmarks/08-kimi-k3.html","configuration":null},{"observationId":"s-vulcanbench-muse-spark-1.2-19-musespark-effort-high-2f58434bbe3d","modelId":"muse-spark-1.2","label":"Not specified","sourceUrl":"https://vulcanbench.com/benchmarks/19-musespark-effort.html","configuration":null},{"observationId":"s-vulcanbench-muse-spark-1.2-19-musespark-effort-low-2f58434bbe3d","modelId":"muse-spark-1.2","label":"Not specified","sourceUrl":"https://vulcanbench.com/benchmarks/19-musespark-effort.html","configuration":null},{"observationId":"s-vulcanbench-muse-spark-1.2-19-musespark-effort-xhigh-2f58434bbe3d","modelId":"muse-spark-1.2","label":"Not specified","sourceUrl":"https://vulcanbench.com/benchmarks/19-musespark-effort.html","configuration":null},{"observationId":"s-vulcanbench-qwen-3.8-max-12-qwen38-max-low-2f58434bbe3d","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://vulcanbench.com/benchmarks/12-qwen38-max.html","configuration":null},{"observationId":"s-vulcanbench-qwen-3.8-max-12-qwen38-max-medium-2f58434bbe3d","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://vulcanbench.com/benchmarks/12-qwen38-max.html","configuration":null},{"observationId":"s-vulcanbench-qwen-3.8-max-12-qwen38-max-xhigh-2f58434bbe3d","modelId":"qwen-3.8-max","label":"Not specified","sourceUrl":"https://vulcanbench.com/benchmarks/12-qwen38-max.html","configuration":null},{"observationId":"s-vulcanbench-qwen3.8-27b-17-qwen38-27b-effort-low-2f58434bbe3d","modelId":"qwen3.8-27b","label":"Not specified","sourceUrl":"https://vulcanbench.com/benchmarks/17-qwen38-27b-effort.html","configuration":null},{"observationId":"s-vulcanbench-qwen3.8-27b-17-qwen38-27b-effort-medium-2f58434bbe3d","modelId":"qwen3.8-27b","label":"Not specified","sourceUrl":"https://vulcanbench.com/benchmarks/17-qwen38-27b-effort.html","configuration":null},{"observationId":"s-vulcanbench-qwen3.8-27b-17-qwen38-27b-effort-xhigh-2f58434bbe3d","modelId":"qwen3.8-27b","label":"Not specified","sourceUrl":"https://vulcanbench.com/benchmarks/17-qwen38-27b-effort.html","configuration":null}]}