{
    "metadata": {
        "benchmark": "SRE Bench",
        "slug": "srebench",
        "description": "Can AI agents reverse engineer real-world, contamination-free binaries?",
        "benchmark_id": "sre_bench",
        "family": "sre_bench",
        "version": "1",
        "updated": "2026-10-04",
        "dataset_type": "private",
        "industry": "cyber",
        "tasks": {
            "overall": "Overall",
            "solved": "Fully Solved",
            "partial": "Capability Score"
        },
        "models": [
            "anthropic/claude-fable-5-1",
            "anthropic/claude-opus-5",
            "anthropic/claude-opus-5-5",
            "anthropic/claude-opus-5-5-range",
            "anthropic/claude-sonnet-5",
            "anthropic/claude-sonnet-5-5",
            "deepseek/deepseek-v4-pro-0813",
            "deepseek/deepseek-v4.1-flash",
            "google/gemini-3.1-pro-preview",
            "google/gemini-3.7-flash",
            "google/gemini-3.8-flash",
            "google/gemini-4-argon",
            "grok/grok-4.5",
            "grok/grok-4.7",
            "meta/muse_spark_1_2",
            "meta/muse_spark_1_3_max",
            "openai/gpt-5.5",
            "openai/gpt-5.6-luna",
            "openai/gpt-5.6-sol",
            "openai/gpt-5.6-terra",
            "openai/gpt-6-astra",
            "openai/gpt-6-luna",
            "openai/gpt-6-sol",
            "openai/gpt-6.1-sol",
            "openai/gpt-daybreak-blue",
            "tencent/hy4-preview",
            "thinkingmachines/inkling",
            "xiaomi/mimo-v2.6-flash",
            "xiaomi/mimo-v2.6-pro",
            "zai/glm-5.2",
            "zai/glm-5.3",
            "zai/glm-5.3-flash"
        ],
        "partners": [],
        "showBadge": false,
        "visible": true,
        "use_cost_per_test": true,
        "runner": "external",
        "mode": "agentic",
        "archived": false,
        "partner": false,
        "total_models": 32
    },
    "tasks": {
        "overall": {
            "openai/gpt-6-astra": {
                "accuracy": 56.87,
                "tie_breaker_score": 63.868,
                "latency": 1070.676,
                "stderr": 3.066,
                "cost_per_test": 10.238062,
                "token_totals": {
                    "input_tokens": 1382583670,
                    "output_tokens": 8281646,
                    "reasoning_tokens": 3776497,
                    "cache_read_tokens": 1298071568,
                    "cache_write_tokens": null
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-6.1-sol": {
                "accuracy": 50.763,
                "tie_breaker_score": 56.298,
                "latency": 1952.046,
                "stderr": 3.095,
                "cost_per_test": 2.685175,
                "token_totals": {
                    "input_tokens": 2060874823,
                    "output_tokens": 11701048,
                    "reasoning_tokens": 6536861,
                    "cache_read_tokens": 2003867049,
                    "cache_write_tokens": 56528756
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-4-argon": {
                "accuracy": 44.275,
                "tie_breaker_score": 57.824,
                "latency": 5695.712,
                "stderr": 3.075,
                "cost_per_test": 30.451343,
                "token_totals": {
                    "input_tokens": 18511814639,
                    "output_tokens": 43486854,
                    "reasoning_tokens": 28271875,
                    "cache_read_tokens": 17516454239,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-5-5": {
                "accuracy": 33.588,
                "tie_breaker_score": 53.562,
                "latency": 5795.029,
                "stderr": 2.923,
                "cost_per_test": 34.654445,
                "token_totals": {
                    "input_tokens": 13829392481,
                    "output_tokens": 60087914,
                    "reasoning_tokens": 39110972,
                    "cache_read_tokens": 13577780702,
                    "cache_write_tokens": 251504746
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-daybreak-blue": {
                "accuracy": 31.679,
                "tie_breaker_score": 58.079,
                "latency": 10424.242,
                "stderr": 2.88,
                "cost_per_test": 54.338843,
                "token_totals": {
                    "input_tokens": 13946863921,
                    "output_tokens": 37448669,
                    "reasoning_tokens": 23681070,
                    "cache_read_tokens": 13831167742,
                    "cache_write_tokens": 115224484
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.6-sol": {
                "accuracy": 30.534,
                "tie_breaker_score": 59.542,
                "latency": 5285.426,
                "stderr": 2.851,
                "cost_per_test": 42.521152,
                "token_totals": {
                    "input_tokens": 15309842096,
                    "output_tokens": 29542116,
                    "reasoning_tokens": 13735453,
                    "cache_read_tokens": 14807800308,
                    "cache_write_tokens": null
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-6-sol": {
                "accuracy": 30.534,
                "tie_breaker_score": 57.252,
                "latency": 3807.782,
                "stderr": 2.851,
                "cost_per_test": 20.651048,
                "token_totals": {
                    "input_tokens": 13521569335,
                    "output_tokens": 35917420,
                    "reasoning_tokens": 22503283,
                    "cache_read_tokens": 13407848991,
                    "cache_write_tokens": 113335637
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-sonnet-5-5": {
                "accuracy": 30.153,
                "tie_breaker_score": 51.59,
                "latency": 6881.468,
                "stderr": 2.841,
                "cost_per_test": 26.52067,
                "token_totals": {
                    "input_tokens": 20607388484,
                    "output_tokens": 87834209,
                    "reasoning_tokens": 63360816,
                    "cache_read_tokens": 20257629467,
                    "cache_write_tokens": 349632633
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-fable-5-1": {
                "accuracy": 22.901,
                "tie_breaker_score": 36.641,
                "latency": 7375.578,
                "stderr": 2.601,
                "cost_per_test": 32.231825,
                "token_totals": {
                    "input_tokens": 9153057446,
                    "output_tokens": 52495018,
                    "reasoning_tokens": 35485733,
                    "cache_read_tokens": 8923282975,
                    "cache_write_tokens": 229697119
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-5-5-range": {
                "accuracy": 21.756,
                "tie_breaker_score": 42.621,
                "latency": 12117.679,
                "stderr": 2.554,
                "cost_per_test": 33.133504,
                "token_totals": {
                    "input_tokens": 10299431517,
                    "output_tokens": 53649193,
                    "reasoning_tokens": 35272730,
                    "cache_read_tokens": 9858505552,
                    "cache_write_tokens": 440838873
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-5": {
                "accuracy": 12.214,
                "tie_breaker_score": 31.107,
                "latency": 4858.213,
                "stderr": 2.027,
                "cost_per_test": 23.633213,
                "token_totals": {
                    "input_tokens": 7191259372,
                    "output_tokens": 50258561,
                    "reasoning_tokens": 34672581,
                    "cache_read_tokens": 6982893726,
                    "cache_write_tokens": 208290816
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.8-flash": {
                "accuracy": 11.069,
                "tie_breaker_score": 19.656,
                "latency": 8853.856,
                "stderr": 1.942,
                "cost_per_test": 27.459572,
                "token_totals": {
                    "input_tokens": 37086688061,
                    "output_tokens": 61375519,
                    "reasoning_tokens": 34869435,
                    "cache_read_tokens": 36219215322,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.6-terra": {
                "accuracy": 9.542,
                "tie_breaker_score": 26.845,
                "latency": 11800.798,
                "stderr": 1.819,
                "cost_per_test": 30.855656,
                "token_totals": {
                    "input_tokens": 19277033725,
                    "output_tokens": 38557051,
                    "reasoning_tokens": 21920260,
                    "cache_read_tokens": 19119371594,
                    "cache_write_tokens": 156882339
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-5.3": {
                "accuracy": 9.16,
                "tie_breaker_score": 19.529,
                "latency": 10089.867,
                "stderr": 1.786,
                "cost_per_test": 27.465302,
                "token_totals": {
                    "input_tokens": 16368358661,
                    "output_tokens": 71364631,
                    "reasoning_tokens": 54858524,
                    "cache_read_tokens": 14064734628,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "google/gemini-3.7-flash": {
                "accuracy": 4.58,
                "tie_breaker_score": 14.186,
                "latency": 6606.13,
                "stderr": 1.294,
                "cost_per_test": 21.474171,
                "token_totals": {
                    "input_tokens": 22760467611,
                    "output_tokens": 50815880,
                    "reasoning_tokens": 20893173,
                    "cache_read_tokens": 21833622412,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "xiaomi/mimo-v2.6-flash": {
                "accuracy": 4.198,
                "tie_breaker_score": 15.903,
                "latency": 12121.727,
                "stderr": 1.241,
                "cost_per_test": 0.395481,
                "token_totals": {
                    "input_tokens": 19413117046,
                    "output_tokens": 88511668,
                    "reasoning_tokens": 66586385,
                    "cache_read_tokens": 19232447040,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "grok/grok-4.7": {
                "accuracy": 3.817,
                "tie_breaker_score": 17.557,
                "latency": 6042.744,
                "stderr": 1.186,
                "cost_per_test": 64.767258,
                "token_totals": {
                    "input_tokens": 13965385351,
                    "output_tokens": 52446537,
                    "reasoning_tokens": 27272060,
                    "cache_read_tokens": 12714802816,
                    "cache_write_tokens": 0
                },
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 3.817,
                "tie_breaker_score": 17.048,
                "latency": 2787.997,
                "stderr": 1.186,
                "cost_per_test": 17.773692,
                "token_totals": {
                    "input_tokens": 5148417224,
                    "output_tokens": 16855383,
                    "reasoning_tokens": 8349676,
                    "cache_read_tokens": 5009534976,
                    "cache_write_tokens": null
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "meta/muse_spark_1_3_max": {
                "accuracy": 3.817,
                "tie_breaker_score": 12.277,
                "latency": 6407.234,
                "stderr": 1.186,
                "cost_per_test": 25.861222,
                "token_totals": {
                    "input_tokens": 24757888036,
                    "output_tokens": 99664531,
                    "reasoning_tokens": 68826353,
                    "cache_read_tokens": 22359358252,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "xiaomi/mimo-v2.6-pro": {
                "accuracy": 3.053,
                "tie_breaker_score": 15.84,
                "latency": 8562.864,
                "stderr": 1.065,
                "cost_per_test": 0.53277,
                "token_totals": {
                    "input_tokens": 8272484434,
                    "output_tokens": 64889760,
                    "reasoning_tokens": 51541453,
                    "cache_read_tokens": 8147784320,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "openai/gpt-6-luna": {
                "accuracy": 2.672,
                "tie_breaker_score": 11.132,
                "latency": 4871.097,
                "stderr": 0.998,
                "cost_per_test": 1.550056,
                "token_totals": {
                    "input_tokens": 17705307873,
                    "output_tokens": 76403083,
                    "reasoning_tokens": 59461162,
                    "cache_read_tokens": 17550144252,
                    "cache_write_tokens": 154857879
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "tencent/hy4-preview": {
                "accuracy": 2.29,
                "tie_breaker_score": 10.814,
                "latency": 5996.674,
                "stderr": 0.926,
                "cost_per_test": 4.902269,
                "token_totals": {
                    "input_tokens": 6869685212,
                    "output_tokens": 48702159,
                    "reasoning_tokens": 37860929,
                    "cache_read_tokens": 5769840896,
                    "cache_write_tokens": 0
                },
                "temperature": 0.9,
                "top_p": 1.0,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Tencent",
                "harness": null
            },
            "deepseek/deepseek-v4-pro-0813": {
                "accuracy": 1.908,
                "tie_breaker_score": 10.56,
                "latency": 8425.728,
                "stderr": 0.847,
                "cost_per_test": 5.579756,
                "token_totals": {
                    "input_tokens": 24734059435,
                    "output_tokens": 67088556,
                    "reasoning_tokens": 39595412,
                    "cache_read_tokens": 24649477248,
                    "cache_write_tokens": null
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "openai/gpt-5.6-luna": {
                "accuracy": 1.527,
                "tie_breaker_score": 6.361,
                "latency": 6114.391,
                "stderr": 0.759,
                "cost_per_test": 3.439085,
                "token_totals": {
                    "input_tokens": 21322002349,
                    "output_tokens": 46327026,
                    "reasoning_tokens": 19502101,
                    "cache_read_tokens": 21181004568,
                    "cache_write_tokens": 140586525
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "deepseek/deepseek-v4.1-flash": {
                "accuracy": 0.763,
                "tie_breaker_score": 14.249,
                "latency": 3663.235,
                "stderr": 0.539,
                "cost_per_test": 0.548583,
                "token_totals": {
                    "input_tokens": 10296939941,
                    "output_tokens": 55771490,
                    "reasoning_tokens": 41799913,
                    "cache_read_tokens": 10245847296,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "zai/glm-5.3-flash": {
                "accuracy": 0.763,
                "tie_breaker_score": 6.489,
                "latency": 11693.322,
                "stderr": 0.539,
                "cost_per_test": 2.290191,
                "token_totals": {
                    "input_tokens": 9240075553,
                    "output_tokens": 43245596,
                    "reasoning_tokens": 31335106,
                    "cache_read_tokens": 6730033152,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "grok/grok-4.5": {
                "accuracy": 0.763,
                "tie_breaker_score": 6.298,
                "latency": 3939.029,
                "stderr": 0.539,
                "cost_per_test": 13.415223,
                "token_totals": {
                    "input_tokens": 3851118391,
                    "output_tokens": 30954518,
                    "reasoning_tokens": 7439099,
                    "cache_read_tokens": 3517266432,
                    "cache_write_tokens": null
                },
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-sonnet-5": {
                "accuracy": 0.382,
                "tie_breaker_score": 7.697,
                "latency": 13984.551,
                "stderr": 0.382,
                "cost_per_test": 18.493954,
                "token_totals": {
                    "input_tokens": 14613737737,
                    "output_tokens": 78491897,
                    "reasoning_tokens": 60616567,
                    "cache_read_tokens": 14119042272,
                    "cache_write_tokens": 494595347
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "zai/glm-5.2": {
                "accuracy": 0.0,
                "tie_breaker_score": 3.435,
                "latency": 8960.277,
                "stderr": 0.0,
                "cost_per_test": 26.575264,
                "token_totals": {
                    "input_tokens": 6115441036,
                    "output_tokens": 53703363,
                    "reasoning_tokens": null,
                    "cache_read_tokens": 1456502354,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "meta/muse_spark_1_2": {
                "accuracy": 0.0,
                "tie_breaker_score": 1.463,
                "latency": 5986.803,
                "stderr": 0.0,
                "cost_per_test": 23.427477,
                "token_totals": {
                    "input_tokens": 10247621050,
                    "output_tokens": 51252763,
                    "reasoning_tokens": 28954284,
                    "cache_read_tokens": 6263046809,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 0.0,
                "tie_breaker_score": 0.954,
                "latency": 3803.735,
                "stderr": 0.0,
                "cost_per_test": 11.336961,
                "token_totals": {
                    "input_tokens": 5876432996,
                    "output_tokens": 22062729,
                    "reasoning_tokens": 15158240,
                    "cache_read_tokens": 5490229595,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "thinkingmachines/inkling": {
                "accuracy": 0.0,
                "tie_breaker_score": 0.254,
                "latency": 1718.686,
                "stderr": 0.0,
                "cost_per_test": 2.225036,
                "token_totals": {
                    "input_tokens": 2816857553,
                    "output_tokens": 16906747,
                    "reasoning_tokens": null,
                    "cache_read_tokens": 2773940224,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            }
        },
        "solved": {
            "openai/gpt-6-astra": {
                "accuracy": 56.87,
                "latency": 1070.676,
                "stderr": 3.066,
                "cost_per_test": 10.238062,
                "token_totals": {
                    "input_tokens": 1382583670,
                    "output_tokens": 8281646,
                    "reasoning_tokens": 3776497,
                    "cache_read_tokens": 1298071568,
                    "cache_write_tokens": null
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-6.1-sol": {
                "accuracy": 50.763,
                "latency": 1952.046,
                "stderr": 3.095,
                "cost_per_test": 2.685175,
                "token_totals": {
                    "input_tokens": 2060874823,
                    "output_tokens": 11701048,
                    "reasoning_tokens": 6536861,
                    "cache_read_tokens": 2003867049,
                    "cache_write_tokens": 56528756
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-4-argon": {
                "accuracy": 44.275,
                "latency": 5695.712,
                "stderr": 3.075,
                "cost_per_test": 30.451343,
                "token_totals": {
                    "input_tokens": 18511814639,
                    "output_tokens": 43486854,
                    "reasoning_tokens": 28271875,
                    "cache_read_tokens": 17516454239,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-5-5": {
                "accuracy": 33.588,
                "latency": 5795.029,
                "stderr": 2.923,
                "cost_per_test": 34.654445,
                "token_totals": {
                    "input_tokens": 13829392481,
                    "output_tokens": 60087914,
                    "reasoning_tokens": 39110972,
                    "cache_read_tokens": 13577780702,
                    "cache_write_tokens": 251504746
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-daybreak-blue": {
                "accuracy": 31.679,
                "latency": 10424.242,
                "stderr": 2.88,
                "cost_per_test": 54.338843,
                "token_totals": {
                    "input_tokens": 13946863921,
                    "output_tokens": 37448669,
                    "reasoning_tokens": 23681070,
                    "cache_read_tokens": 13831167742,
                    "cache_write_tokens": 115224484
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-6-sol": {
                "accuracy": 30.534,
                "latency": 3807.782,
                "stderr": 2.851,
                "cost_per_test": 20.651048,
                "token_totals": {
                    "input_tokens": 13521569335,
                    "output_tokens": 35917420,
                    "reasoning_tokens": 22503283,
                    "cache_read_tokens": 13407848991,
                    "cache_write_tokens": 113335637
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.6-sol": {
                "accuracy": 30.534,
                "latency": 5285.426,
                "stderr": 2.851,
                "cost_per_test": 42.521152,
                "token_totals": {
                    "input_tokens": 15309842096,
                    "output_tokens": 29542116,
                    "reasoning_tokens": 13735453,
                    "cache_read_tokens": 14807800308,
                    "cache_write_tokens": null
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-sonnet-5-5": {
                "accuracy": 30.153,
                "latency": 6881.468,
                "stderr": 2.841,
                "cost_per_test": 26.52067,
                "token_totals": {
                    "input_tokens": 20607388484,
                    "output_tokens": 87834209,
                    "reasoning_tokens": 63360816,
                    "cache_read_tokens": 20257629467,
                    "cache_write_tokens": 349632633
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-fable-5-1": {
                "accuracy": 22.901,
                "latency": 7375.578,
                "stderr": 2.601,
                "cost_per_test": 32.231825,
                "token_totals": {
                    "input_tokens": 9153057446,
                    "output_tokens": 52495018,
                    "reasoning_tokens": 35485733,
                    "cache_read_tokens": 8923282975,
                    "cache_write_tokens": 229697119
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-5-5-range": {
                "accuracy": 21.756,
                "latency": 12117.679,
                "stderr": 2.554,
                "cost_per_test": 33.133504,
                "token_totals": {
                    "input_tokens": 10299431517,
                    "output_tokens": 53649193,
                    "reasoning_tokens": 35272730,
                    "cache_read_tokens": 9858505552,
                    "cache_write_tokens": 440838873
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-5": {
                "accuracy": 12.214,
                "latency": 4858.213,
                "stderr": 2.027,
                "cost_per_test": 23.633213,
                "token_totals": {
                    "input_tokens": 7191259372,
                    "output_tokens": 50258561,
                    "reasoning_tokens": 34672581,
                    "cache_read_tokens": 6982893726,
                    "cache_write_tokens": 208290816
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.8-flash": {
                "accuracy": 11.069,
                "latency": 8853.856,
                "stderr": 1.942,
                "cost_per_test": 27.459572,
                "token_totals": {
                    "input_tokens": 37086688061,
                    "output_tokens": 61375519,
                    "reasoning_tokens": 34869435,
                    "cache_read_tokens": 36219215322,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.6-terra": {
                "accuracy": 9.542,
                "latency": 11800.798,
                "stderr": 1.819,
                "cost_per_test": 30.855656,
                "token_totals": {
                    "input_tokens": 19277033725,
                    "output_tokens": 38557051,
                    "reasoning_tokens": 21920260,
                    "cache_read_tokens": 19119371594,
                    "cache_write_tokens": 156882339
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-5.3": {
                "accuracy": 9.16,
                "latency": 10089.867,
                "stderr": 1.786,
                "cost_per_test": 27.465302,
                "token_totals": {
                    "input_tokens": 16368358661,
                    "output_tokens": 71364631,
                    "reasoning_tokens": 54858524,
                    "cache_read_tokens": 14064734628,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "google/gemini-3.7-flash": {
                "accuracy": 4.58,
                "latency": 6606.13,
                "stderr": 1.294,
                "cost_per_test": 21.474171,
                "token_totals": {
                    "input_tokens": 22760467611,
                    "output_tokens": 50815880,
                    "reasoning_tokens": 20893173,
                    "cache_read_tokens": 21833622412,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "xiaomi/mimo-v2.6-flash": {
                "accuracy": 4.198,
                "latency": 12121.727,
                "stderr": 1.241,
                "cost_per_test": 0.395481,
                "token_totals": {
                    "input_tokens": 19413117046,
                    "output_tokens": 88511668,
                    "reasoning_tokens": 66586385,
                    "cache_read_tokens": 19232447040,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 3.817,
                "latency": 2787.997,
                "stderr": 1.186,
                "cost_per_test": 17.773692,
                "token_totals": {
                    "input_tokens": 5148417224,
                    "output_tokens": 16855383,
                    "reasoning_tokens": 8349676,
                    "cache_read_tokens": 5009534976,
                    "cache_write_tokens": null
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4.7": {
                "accuracy": 3.817,
                "latency": 6042.744,
                "stderr": 1.186,
                "cost_per_test": 64.767258,
                "token_totals": {
                    "input_tokens": 13965385351,
                    "output_tokens": 52446537,
                    "reasoning_tokens": 27272060,
                    "cache_read_tokens": 12714802816,
                    "cache_write_tokens": 0
                },
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "meta/muse_spark_1_3_max": {
                "accuracy": 3.817,
                "latency": 6407.234,
                "stderr": 1.186,
                "cost_per_test": 25.861222,
                "token_totals": {
                    "input_tokens": 24757888036,
                    "output_tokens": 99664531,
                    "reasoning_tokens": 68826353,
                    "cache_read_tokens": 22359358252,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "xiaomi/mimo-v2.6-pro": {
                "accuracy": 3.053,
                "latency": 8562.864,
                "stderr": 1.065,
                "cost_per_test": 0.53277,
                "token_totals": {
                    "input_tokens": 8272484434,
                    "output_tokens": 64889760,
                    "reasoning_tokens": 51541453,
                    "cache_read_tokens": 8147784320,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "openai/gpt-6-luna": {
                "accuracy": 2.672,
                "latency": 4871.097,
                "stderr": 0.998,
                "cost_per_test": 1.550056,
                "token_totals": {
                    "input_tokens": 17705307873,
                    "output_tokens": 76403083,
                    "reasoning_tokens": 59461162,
                    "cache_read_tokens": 17550144252,
                    "cache_write_tokens": 154857879
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "tencent/hy4-preview": {
                "accuracy": 2.29,
                "latency": 5996.674,
                "stderr": 0.926,
                "cost_per_test": 4.902269,
                "token_totals": {
                    "input_tokens": 6869685212,
                    "output_tokens": 48702159,
                    "reasoning_tokens": 37860929,
                    "cache_read_tokens": 5769840896,
                    "cache_write_tokens": 0
                },
                "temperature": 0.9,
                "top_p": 1.0,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Tencent",
                "harness": null
            },
            "deepseek/deepseek-v4-pro-0813": {
                "accuracy": 1.908,
                "latency": 8425.728,
                "stderr": 0.847,
                "cost_per_test": 5.579756,
                "token_totals": {
                    "input_tokens": 24734059435,
                    "output_tokens": 67088556,
                    "reasoning_tokens": 39595412,
                    "cache_read_tokens": 24649477248,
                    "cache_write_tokens": null
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "openai/gpt-5.6-luna": {
                "accuracy": 1.527,
                "latency": 6114.391,
                "stderr": 0.759,
                "cost_per_test": 3.439085,
                "token_totals": {
                    "input_tokens": 21322002349,
                    "output_tokens": 46327026,
                    "reasoning_tokens": 19502101,
                    "cache_read_tokens": 21181004568,
                    "cache_write_tokens": 140586525
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "deepseek/deepseek-v4.1-flash": {
                "accuracy": 0.763,
                "latency": 3663.235,
                "stderr": 0.539,
                "cost_per_test": 0.548583,
                "token_totals": {
                    "input_tokens": 10296939941,
                    "output_tokens": 55771490,
                    "reasoning_tokens": 41799913,
                    "cache_read_tokens": 10245847296,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "grok/grok-4.5": {
                "accuracy": 0.763,
                "latency": 3939.029,
                "stderr": 0.539,
                "cost_per_test": 13.415223,
                "token_totals": {
                    "input_tokens": 3851118391,
                    "output_tokens": 30954518,
                    "reasoning_tokens": 7439099,
                    "cache_read_tokens": 3517266432,
                    "cache_write_tokens": null
                },
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "zai/glm-5.3-flash": {
                "accuracy": 0.763,
                "latency": 11693.322,
                "stderr": 0.539,
                "cost_per_test": 2.290191,
                "token_totals": {
                    "input_tokens": 9240075553,
                    "output_tokens": 43245596,
                    "reasoning_tokens": 31335106,
                    "cache_read_tokens": 6730033152,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "anthropic/claude-sonnet-5": {
                "accuracy": 0.382,
                "latency": 13984.551,
                "stderr": 0.382,
                "cost_per_test": 18.493954,
                "token_totals": {
                    "input_tokens": 14613737737,
                    "output_tokens": 78491897,
                    "reasoning_tokens": 60616567,
                    "cache_read_tokens": 14119042272,
                    "cache_write_tokens": 494595347
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "thinkingmachines/inkling": {
                "accuracy": 0.0,
                "latency": 1718.686,
                "stderr": 0.0,
                "cost_per_test": 2.225036,
                "token_totals": {
                    "input_tokens": 2816857553,
                    "output_tokens": 16906747,
                    "reasoning_tokens": null,
                    "cache_read_tokens": 2773940224,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 0.0,
                "latency": 3803.735,
                "stderr": 0.0,
                "cost_per_test": 11.336961,
                "token_totals": {
                    "input_tokens": 5876432996,
                    "output_tokens": 22062729,
                    "reasoning_tokens": 15158240,
                    "cache_read_tokens": 5490229595,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "meta/muse_spark_1_2": {
                "accuracy": 0.0,
                "latency": 5986.803,
                "stderr": 0.0,
                "cost_per_test": 23.427477,
                "token_totals": {
                    "input_tokens": 10247621050,
                    "output_tokens": 51252763,
                    "reasoning_tokens": 28954284,
                    "cache_read_tokens": 6263046809,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "zai/glm-5.2": {
                "accuracy": 0.0,
                "latency": 8960.277,
                "stderr": 0.0,
                "cost_per_test": 26.575264,
                "token_totals": {
                    "input_tokens": 6115441036,
                    "output_tokens": 53703363,
                    "reasoning_tokens": null,
                    "cache_read_tokens": 1456502354,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            }
        },
        "partial": {
            "openai/gpt-6-astra": {
                "accuracy": 63.868,
                "latency": 1070.676,
                "stderr": 2.859,
                "cost_per_test": 10.238062,
                "token_totals": {
                    "input_tokens": 1382583670,
                    "output_tokens": 8281646,
                    "reasoning_tokens": 3776497,
                    "cache_read_tokens": 1298071568,
                    "cache_write_tokens": null
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.6-sol": {
                "accuracy": 59.542,
                "latency": 5285.426,
                "stderr": 2.289,
                "cost_per_test": 42.521152,
                "token_totals": {
                    "input_tokens": 15309842096,
                    "output_tokens": 29542116,
                    "reasoning_tokens": 13735453,
                    "cache_read_tokens": 14807800308,
                    "cache_write_tokens": null
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-daybreak-blue": {
                "accuracy": 58.079,
                "latency": 10424.242,
                "stderr": 2.367,
                "cost_per_test": 54.338843,
                "token_totals": {
                    "input_tokens": 13946863921,
                    "output_tokens": 37448669,
                    "reasoning_tokens": 23681070,
                    "cache_read_tokens": 13831167742,
                    "cache_write_tokens": 115224484
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-4-argon": {
                "accuracy": 57.824,
                "latency": 5695.712,
                "stderr": 2.878,
                "cost_per_test": 30.451343,
                "token_totals": {
                    "input_tokens": 18511814639,
                    "output_tokens": 43486854,
                    "reasoning_tokens": 28271875,
                    "cache_read_tokens": 17516454239,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-6-sol": {
                "accuracy": 57.252,
                "latency": 3807.782,
                "stderr": 2.349,
                "cost_per_test": 20.651048,
                "token_totals": {
                    "input_tokens": 13521569335,
                    "output_tokens": 35917420,
                    "reasoning_tokens": 22503283,
                    "cache_read_tokens": 13407848991,
                    "cache_write_tokens": 113335637
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-6.1-sol": {
                "accuracy": 56.298,
                "latency": 1952.046,
                "stderr": 2.984,
                "cost_per_test": 2.685175,
                "token_totals": {
                    "input_tokens": 2060874823,
                    "output_tokens": 11701048,
                    "reasoning_tokens": 6536861,
                    "cache_read_tokens": 2003867049,
                    "cache_write_tokens": 56528756
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-5-5": {
                "accuracy": 53.562,
                "latency": 5795.029,
                "stderr": 2.771,
                "cost_per_test": 34.654445,
                "token_totals": {
                    "input_tokens": 13829392481,
                    "output_tokens": 60087914,
                    "reasoning_tokens": 39110972,
                    "cache_read_tokens": 13577780702,
                    "cache_write_tokens": 251504746
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-sonnet-5-5": {
                "accuracy": 51.59,
                "latency": 6881.468,
                "stderr": 2.662,
                "cost_per_test": 26.52067,
                "token_totals": {
                    "input_tokens": 20607388484,
                    "output_tokens": 87834209,
                    "reasoning_tokens": 63360816,
                    "cache_read_tokens": 20257629467,
                    "cache_write_tokens": 349632633
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-5-5-range": {
                "accuracy": 42.621,
                "latency": 12117.679,
                "stderr": 2.575,
                "cost_per_test": 33.133504,
                "token_totals": {
                    "input_tokens": 10299431517,
                    "output_tokens": 53649193,
                    "reasoning_tokens": 35272730,
                    "cache_read_tokens": 9858505552,
                    "cache_write_tokens": 440838873
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-fable-5-1": {
                "accuracy": 36.641,
                "latency": 7375.578,
                "stderr": 2.72,
                "cost_per_test": 32.231825,
                "token_totals": {
                    "input_tokens": 9153057446,
                    "output_tokens": 52495018,
                    "reasoning_tokens": 35485733,
                    "cache_read_tokens": 8923282975,
                    "cache_write_tokens": 229697119
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-5": {
                "accuracy": 31.107,
                "latency": 4858.213,
                "stderr": 2.314,
                "cost_per_test": 23.633213,
                "token_totals": {
                    "input_tokens": 7191259372,
                    "output_tokens": 50258561,
                    "reasoning_tokens": 34672581,
                    "cache_read_tokens": 6982893726,
                    "cache_write_tokens": 208290816
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.6-terra": {
                "accuracy": 26.845,
                "latency": 11800.798,
                "stderr": 2.182,
                "cost_per_test": 30.855656,
                "token_totals": {
                    "input_tokens": 19277033725,
                    "output_tokens": 38557051,
                    "reasoning_tokens": 21920260,
                    "cache_read_tokens": 19119371594,
                    "cache_write_tokens": 156882339
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3.8-flash": {
                "accuracy": 19.656,
                "latency": 8853.856,
                "stderr": 2.087,
                "cost_per_test": 27.459572,
                "token_totals": {
                    "input_tokens": 37086688061,
                    "output_tokens": 61375519,
                    "reasoning_tokens": 34869435,
                    "cache_read_tokens": 36219215322,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "zai/glm-5.3": {
                "accuracy": 19.529,
                "latency": 10089.867,
                "stderr": 2.003,
                "cost_per_test": 27.465302,
                "token_totals": {
                    "input_tokens": 16368358661,
                    "output_tokens": 71364631,
                    "reasoning_tokens": 54858524,
                    "cache_read_tokens": 14064734628,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "grok/grok-4.7": {
                "accuracy": 17.557,
                "latency": 6042.744,
                "stderr": 1.758,
                "cost_per_test": 64.767258,
                "token_totals": {
                    "input_tokens": 13965385351,
                    "output_tokens": 52446537,
                    "reasoning_tokens": 27272060,
                    "cache_read_tokens": 12714802816,
                    "cache_write_tokens": 0
                },
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 17.048,
                "latency": 2787.997,
                "stderr": 1.647,
                "cost_per_test": 17.773692,
                "token_totals": {
                    "input_tokens": 5148417224,
                    "output_tokens": 16855383,
                    "reasoning_tokens": 8349676,
                    "cache_read_tokens": 5009534976,
                    "cache_write_tokens": null
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "xiaomi/mimo-v2.6-flash": {
                "accuracy": 15.903,
                "latency": 12121.727,
                "stderr": 1.752,
                "cost_per_test": 0.395481,
                "token_totals": {
                    "input_tokens": 19413117046,
                    "output_tokens": 88511668,
                    "reasoning_tokens": 66586385,
                    "cache_read_tokens": 19232447040,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "xiaomi/mimo-v2.6-pro": {
                "accuracy": 15.84,
                "latency": 8562.864,
                "stderr": 1.696,
                "cost_per_test": 0.53277,
                "token_totals": {
                    "input_tokens": 8272484434,
                    "output_tokens": 64889760,
                    "reasoning_tokens": 51541453,
                    "cache_read_tokens": 8147784320,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "deepseek/deepseek-v4.1-flash": {
                "accuracy": 14.249,
                "latency": 3663.235,
                "stderr": 1.479,
                "cost_per_test": 0.548583,
                "token_totals": {
                    "input_tokens": 10296939941,
                    "output_tokens": 55771490,
                    "reasoning_tokens": 41799913,
                    "cache_read_tokens": 10245847296,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "google/gemini-3.7-flash": {
                "accuracy": 14.186,
                "latency": 6606.13,
                "stderr": 1.712,
                "cost_per_test": 21.474171,
                "token_totals": {
                    "input_tokens": 22760467611,
                    "output_tokens": 50815880,
                    "reasoning_tokens": 20893173,
                    "cache_read_tokens": 21833622412,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "meta/muse_spark_1_3_max": {
                "accuracy": 12.277,
                "latency": 6407.234,
                "stderr": 1.578,
                "cost_per_test": 25.861222,
                "token_totals": {
                    "input_tokens": 24757888036,
                    "output_tokens": 99664531,
                    "reasoning_tokens": 68826353,
                    "cache_read_tokens": 22359358252,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "openai/gpt-6-luna": {
                "accuracy": 11.132,
                "latency": 4871.097,
                "stderr": 1.489,
                "cost_per_test": 1.550056,
                "token_totals": {
                    "input_tokens": 17705307873,
                    "output_tokens": 76403083,
                    "reasoning_tokens": 59461162,
                    "cache_read_tokens": 17550144252,
                    "cache_write_tokens": 154857879
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "tencent/hy4-preview": {
                "accuracy": 10.814,
                "latency": 5996.674,
                "stderr": 1.405,
                "cost_per_test": 4.902269,
                "token_totals": {
                    "input_tokens": 6869685212,
                    "output_tokens": 48702159,
                    "reasoning_tokens": 37860929,
                    "cache_read_tokens": 5769840896,
                    "cache_write_tokens": 0
                },
                "temperature": 0.9,
                "top_p": 1.0,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Tencent",
                "harness": null
            },
            "deepseek/deepseek-v4-pro-0813": {
                "accuracy": 10.56,
                "latency": 8425.728,
                "stderr": 1.423,
                "cost_per_test": 5.579756,
                "token_totals": {
                    "input_tokens": 24734059435,
                    "output_tokens": 67088556,
                    "reasoning_tokens": 39595412,
                    "cache_read_tokens": 24649477248,
                    "cache_write_tokens": null
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "anthropic/claude-sonnet-5": {
                "accuracy": 7.697,
                "latency": 13984.551,
                "stderr": 1.072,
                "cost_per_test": 18.493954,
                "token_totals": {
                    "input_tokens": 14613737737,
                    "output_tokens": 78491897,
                    "reasoning_tokens": 60616567,
                    "cache_read_tokens": 14119042272,
                    "cache_write_tokens": 494595347
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "zai/glm-5.3-flash": {
                "accuracy": 6.489,
                "latency": 11693.322,
                "stderr": 1.02,
                "cost_per_test": 2.290191,
                "token_totals": {
                    "input_tokens": 9240075553,
                    "output_tokens": 43245596,
                    "reasoning_tokens": 31335106,
                    "cache_read_tokens": 6730033152,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "openai/gpt-5.6-luna": {
                "accuracy": 6.361,
                "latency": 6114.391,
                "stderr": 1.085,
                "cost_per_test": 3.439085,
                "token_totals": {
                    "input_tokens": 21322002349,
                    "output_tokens": 46327026,
                    "reasoning_tokens": 19502101,
                    "cache_read_tokens": 21181004568,
                    "cache_write_tokens": 140586525
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4.5": {
                "accuracy": 6.298,
                "latency": 3939.029,
                "stderr": 0.991,
                "cost_per_test": 13.415223,
                "token_totals": {
                    "input_tokens": 3851118391,
                    "output_tokens": 30954518,
                    "reasoning_tokens": 7439099,
                    "cache_read_tokens": 3517266432,
                    "cache_write_tokens": null
                },
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "zai/glm-5.2": {
                "accuracy": 3.435,
                "latency": 8960.277,
                "stderr": 0.544,
                "cost_per_test": 26.575264,
                "token_totals": {
                    "input_tokens": 6115441036,
                    "output_tokens": 53703363,
                    "reasoning_tokens": null,
                    "cache_read_tokens": 1456502354,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "meta/muse_spark_1_2": {
                "accuracy": 1.463,
                "latency": 5986.803,
                "stderr": 0.319,
                "cost_per_test": 23.427477,
                "token_totals": {
                    "input_tokens": 10247621050,
                    "output_tokens": 51252763,
                    "reasoning_tokens": 28954284,
                    "cache_read_tokens": 6263046809,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 0.954,
                "latency": 3803.735,
                "stderr": 0.338,
                "cost_per_test": 11.336961,
                "token_totals": {
                    "input_tokens": 5876432996,
                    "output_tokens": 22062729,
                    "reasoning_tokens": 15158240,
                    "cache_read_tokens": 5490229595,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "thinkingmachines/inkling": {
                "accuracy": 0.254,
                "latency": 1718.686,
                "stderr": 0.126,
                "cost_per_test": 2.225036,
                "token_totals": {
                    "input_tokens": 2816857553,
                    "output_tokens": 16906747,
                    "reasoning_tokens": null,
                    "cache_read_tokens": 2773940224,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            }
        }
    }
}