{
    "metadata": {
        "benchmark": "IOI",
        "slug": "ioi",
        "description": "Based on the International Olympiad in Informatics",
        "benchmark_id": "ioi",
        "family": "ioi",
        "version": "2",
        "updated": "2026-10-06",
        "dataset_type": "public",
        "industry": "coding",
        "tasks": {
            "overall": "Overall",
            "ioi2024": "IOI 2024",
            "ioi2025": "IOI 2025",
            "ioi2026": "IOI 2026"
        },
        "models": [
            "alibaba/qwen3.8-27b",
            "alibaba/qwen3.8-max",
            "anthropic/claude-fable-5-1",
            "anthropic/claude-opus-5",
            "anthropic/claude-opus-5-5",
            "anthropic/claude-sonnet-5",
            "anthropic/claude-sonnet-5-5",
            "deepseek/deepseek-v4-flash-0731",
            "deepseek/deepseek-v4-pro-0813",
            "deepseek/deepseek-v4.1-flash",
            "fireworks/ember-1",
            "google/gemini-3.1-pro-preview",
            "google/gemini-3.6-flash",
            "google/gemini-3.7-flash",
            "google/gemini-3.8-flash",
            "google/gemini-4-argon",
            "grok/grok-4.5",
            "grok/grok-4.6",
            "grok/grok-4.7",
            "inception/mercury-2.5",
            "kimi/kimi-k3",
            "meta/muse_spark_1_2",
            "meta/muse_spark_1_3",
            "meta/muse_spark_1_3_max",
            "mistralai/mistral-large-4",
            "openai/gpt-5.3-codex",
            "openai/gpt-5.6-luna",
            "openai/gpt-5.6-sol",
            "openai/gpt-5.6-terra",
            "openai/gpt-6-astra",
            "openai/gpt-6-luna",
            "openai/gpt-6-sol",
            "openai/gpt-6.1-sol",
            "stepfun/step-5-preview",
            "tencent/hy4-preview",
            "thinkingmachines/inkling",
            "thinkingmachines/inkling-small",
            "xiaomi/mimo-v2.6-flash",
            "xiaomi/mimo-v2.6-pro",
            "zai/glm-5.3",
            "zai/glm-5.3-flash"
        ],
        "partners": [],
        "showBadge": false,
        "visible": true,
        "use_cost_per_test": true,
        "runner": "custom",
        "mode": "agentic",
        "archived": false,
        "partner": false,
        "total_models": 41
    },
    "tasks": {
        "overall": {
            "google/gemini-4-argon": {
                "accuracy": 100.0,
                "latency": 1459.27,
                "stderr": 0.0,
                "cost_per_test": 5.307626,
                "token_totals": {
                    "input_tokens": 48678879,
                    "output_tokens": 620036,
                    "reasoning_tokens": 445230,
                    "cache_read_tokens": 46123808,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-6-astra": {
                "accuracy": 100.0,
                "latency": 1899.783,
                "stderr": 0.0,
                "cost_per_test": 6.495051,
                "token_totals": {
                    "input_tokens": 1757046,
                    "output_tokens": 71209,
                    "reasoning_tokens": 40588,
                    "cache_read_tokens": 1590180,
                    "cache_write_tokens": 166635
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-6.1-sol": {
                "accuracy": 96.889,
                "latency": 1890.714,
                "stderr": 3.111,
                "cost_per_test": 1.261598,
                "token_totals": {
                    "input_tokens": 1916776,
                    "output_tokens": 91537,
                    "reasoning_tokens": 61697,
                    "cache_read_tokens": 1711529,
                    "cache_write_tokens": 205001
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-5-5": {
                "accuracy": 95.056,
                "latency": 1035.191,
                "stderr": 4.944,
                "cost_per_test": 5.258418,
                "token_totals": {
                    "input_tokens": 27471087,
                    "output_tokens": 744246,
                    "reasoning_tokens": 630177,
                    "cache_read_tokens": 26284219,
                    "cache_write_tokens": 1186555
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.6-sol": {
                "accuracy": 91.167,
                "latency": 3711.631,
                "stderr": 4.51,
                "cost_per_test": 7.982135,
                "token_totals": {
                    "input_tokens": 4489581,
                    "output_tokens": 156360,
                    "reasoning_tokens": 114595,
                    "cache_read_tokens": 4104315,
                    "cache_write_tokens": 384882
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-fable-5-1": {
                "accuracy": 90.778,
                "latency": 1805.905,
                "stderr": 4.647,
                "cost_per_test": 11.196959,
                "token_totals": {
                    "input_tokens": 36683836,
                    "output_tokens": 982199,
                    "reasoning_tokens": 853725,
                    "cache_read_tokens": 35233518,
                    "cache_write_tokens": 1449894
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.6-terra": {
                "accuracy": 87.611,
                "latency": 4992.838,
                "stderr": 6.531,
                "cost_per_test": 8.664009,
                "token_totals": {
                    "input_tokens": 8455718,
                    "output_tokens": 222620,
                    "reasoning_tokens": 182032,
                    "cache_read_tokens": 8044366,
                    "cache_write_tokens": 410842
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-5": {
                "accuracy": 84.333,
                "latency": 3402.045,
                "stderr": 9.964,
                "cost_per_test": 16.477241,
                "token_totals": {
                    "input_tokens": 96077494,
                    "output_tokens": 1386182,
                    "reasoning_tokens": 1125999,
                    "cache_read_tokens": 93265146,
                    "cache_write_tokens": 2811668
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-sonnet-5-5": {
                "accuracy": 83.056,
                "latency": 2751.084,
                "stderr": 3.737,
                "cost_per_test": 7.524252,
                "token_totals": {
                    "input_tokens": 106806202,
                    "output_tokens": 1512099,
                    "reasoning_tokens": 1288748,
                    "cache_read_tokens": 103054095,
                    "cache_write_tokens": 3751464
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-6-sol": {
                "accuracy": 82.611,
                "latency": 1837.553,
                "stderr": 9.24,
                "cost_per_test": 2.703348,
                "token_totals": {
                    "input_tokens": 4061594,
                    "output_tokens": 147384,
                    "reasoning_tokens": 114768,
                    "cache_read_tokens": 3736261,
                    "cache_write_tokens": 324992
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3.8-max": {
                "accuracy": 68.889,
                "latency": 6398.213,
                "stderr": 5.204,
                "cost_per_test": 9.468852,
                "token_totals": {
                    "input_tokens": 166939284,
                    "output_tokens": 1544590,
                    "reasoning_tokens": 1320041,
                    "cache_read_tokens": 164712576,
                    "cache_write_tokens": 0
                },
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "zai/glm-5.3": {
                "accuracy": 68.444,
                "latency": 6374.465,
                "stderr": 7.476,
                "cost_per_test": 7.67087,
                "token_totals": {
                    "input_tokens": 119980037,
                    "output_tokens": 1912941,
                    "reasoning_tokens": 1774936,
                    "cache_read_tokens": 114647509,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "google/gemini-3.7-flash": {
                "accuracy": 67.833,
                "latency": 1269.021,
                "stderr": 3.606,
                "cost_per_test": 3.398069,
                "token_totals": {
                    "input_tokens": 45709147,
                    "output_tokens": 720057,
                    "reasoning_tokens": 366849,
                    "cache_read_tokens": 39685727,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.6-luna": {
                "accuracy": 61.778,
                "latency": 4175.412,
                "stderr": 11.613,
                "cost_per_test": 0.57961,
                "token_totals": {
                    "input_tokens": 33327725,
                    "output_tokens": 460900,
                    "reasoning_tokens": 354484,
                    "cache_read_tokens": 32366771,
                    "cache_write_tokens": 960008
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "tencent/hy4-preview": {
                "accuracy": 59.333,
                "latency": 3633.295,
                "stderr": 4.596,
                "cost_per_test": 1.354793,
                "token_totals": {
                    "input_tokens": 68751149,
                    "output_tokens": 1215516,
                    "reasoning_tokens": 1077432,
                    "cache_read_tokens": 66250176,
                    "cache_write_tokens": 0
                },
                "temperature": 0.9,
                "top_p": 1.0,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Tencent",
                "harness": null
            },
            "grok/grok-4.7": {
                "accuracy": 57.722,
                "latency": 2808.481,
                "stderr": 1.987,
                "cost_per_test": 12.707908,
                "token_totals": {
                    "input_tokens": 64016094,
                    "output_tokens": 1173746,
                    "reasoning_tokens": 924369,
                    "cache_read_tokens": 57851349,
                    "cache_write_tokens": 0
                },
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-3.8-flash": {
                "accuracy": 56.944,
                "latency": 860.465,
                "stderr": 2.257,
                "cost_per_test": 3.976052,
                "token_totals": {
                    "input_tokens": 80760215,
                    "output_tokens": 904638,
                    "reasoning_tokens": 577481,
                    "cache_read_tokens": 77087994,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "meta/muse_spark_1_3_max": {
                "accuracy": 56.556,
                "latency": 1737.996,
                "stderr": 2.517,
                "cost_per_test": 1.729661,
                "token_totals": {
                    "input_tokens": 24731459,
                    "output_tokens": 1199568,
                    "reasoning_tokens": 1063682,
                    "cache_read_tokens": 23361401,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "openai/gpt-6-luna": {
                "accuracy": 55.556,
                "latency": 2147.141,
                "stderr": 8.866,
                "cost_per_test": 0.199412,
                "token_totals": {
                    "input_tokens": 23476341,
                    "output_tokens": 697861,
                    "reasoning_tokens": 642789,
                    "cache_read_tokens": 22466950,
                    "cache_write_tokens": 1008870
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.3-codex": {
                "accuracy": 53.833,
                "latency": 1791.685,
                "stderr": 6.312,
                "cost_per_test": 2.68083,
                "token_totals": {
                    "input_tokens": 22315885,
                    "output_tokens": 552013,
                    "reasoning_tokens": 420622,
                    "cache_read_tokens": 21352960,
                    "cache_write_tokens": 0
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-5.3-flash": {
                "accuracy": 52.5,
                "latency": 5548.524,
                "stderr": 2.619,
                "cost_per_test": 0.429263,
                "token_totals": {
                    "input_tokens": 105605694,
                    "output_tokens": 1564858,
                    "reasoning_tokens": 1449504,
                    "cache_read_tokens": 99300651,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 51.833,
                "latency": 722.637,
                "stderr": 2.589,
                "cost_per_test": 1.25149,
                "token_totals": {
                    "input_tokens": 7336646,
                    "output_tokens": 321302,
                    "reasoning_tokens": 262939,
                    "cache_read_tokens": 6122209,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "deepseek/deepseek-v4-pro-0813": {
                "accuracy": 51.611,
                "latency": 4045.972,
                "stderr": 2.4,
                "cost_per_test": 2.151589,
                "token_totals": {
                    "input_tokens": 126847094,
                    "output_tokens": 1737645,
                    "reasoning_tokens": 1498891,
                    "cache_read_tokens": 126531285,
                    "cache_write_tokens": 0
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "kimi/kimi-k3": {
                "accuracy": 48.944,
                "latency": 15508.583,
                "stderr": 9.82,
                "cost_per_test": 14.165034,
                "token_totals": {
                    "input_tokens": 128652013,
                    "output_tokens": 2047314,
                    "reasoning_tokens": 1796826,
                    "cache_read_tokens": 122842795,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "xiaomi/mimo-v2.6-flash": {
                "accuracy": 47.667,
                "latency": 6521.606,
                "stderr": 4.785,
                "cost_per_test": 0.215104,
                "token_totals": {
                    "input_tokens": 123657830,
                    "output_tokens": 1422082,
                    "reasoning_tokens": 1254795,
                    "cache_read_tokens": 121214869,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "grok/grok-4.6": {
                "accuracy": 47.611,
                "latency": 14233.553,
                "stderr": 2.091,
                "cost_per_test": 8.183386,
                "token_totals": {
                    "input_tokens": 28135041,
                    "output_tokens": 4997524,
                    "reasoning_tokens": 4640283,
                    "cache_read_tokens": 24831232,
                    "cache_write_tokens": 0
                },
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "fireworks/ember-1": {
                "accuracy": 46.333,
                "latency": 6401.12,
                "stderr": 9.163,
                "cost_per_test": 24.458496,
                "token_totals": {
                    "input_tokens": 334918914,
                    "output_tokens": 1829918,
                    "reasoning_tokens": 1524591,
                    "cache_read_tokens": 327984922,
                    "cache_write_tokens": 0
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "mistralai/mistral-large-4": {
                "accuracy": 45.278,
                "latency": 5023.763,
                "stderr": 3.783,
                "cost_per_test": 5.392691,
                "token_totals": {
                    "input_tokens": 72592958,
                    "output_tokens": 2887943,
                    "reasoning_tokens": 0,
                    "cache_read_tokens": 64296619,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "anthropic/claude-sonnet-5": {
                "accuracy": 45.0,
                "latency": 3583.884,
                "stderr": 2.754,
                "cost_per_test": 12.889951,
                "token_totals": {
                    "input_tokens": 227584620,
                    "output_tokens": 2084584,
                    "reasoning_tokens": 1802695,
                    "cache_read_tokens": 222811755,
                    "cache_write_tokens": 4771571
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "meta/muse_spark_1_3": {
                "accuracy": 43.944,
                "latency": 1791.133,
                "stderr": 1.448,
                "cost_per_test": 2.360741,
                "token_totals": {
                    "input_tokens": 28367260,
                    "output_tokens": 1984488,
                    "reasoning_tokens": 1867961,
                    "cache_read_tokens": 27101376,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "grok/grok-4.5": {
                "accuracy": 40.556,
                "latency": 5194.818,
                "stderr": 3.951,
                "cost_per_test": 5.637814,
                "token_totals": {
                    "input_tokens": 37239163,
                    "output_tokens": 1518547,
                    "reasoning_tokens": 965315,
                    "cache_read_tokens": 32620288,
                    "cache_write_tokens": 0
                },
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "deepseek/deepseek-v4.1-flash": {
                "accuracy": 40.278,
                "latency": 1109.577,
                "stderr": 2.557,
                "cost_per_test": 0.24255,
                "token_totals": {
                    "input_tokens": 39221134,
                    "output_tokens": 970177,
                    "reasoning_tokens": 853932,
                    "cache_read_tokens": 39069653,
                    "cache_write_tokens": 0
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "xiaomi/mimo-v2.6-pro": {
                "accuracy": 39.333,
                "latency": 8782.516,
                "stderr": 2.406,
                "cost_per_test": 0.835497,
                "token_totals": {
                    "input_tokens": 39729593,
                    "output_tokens": 2885122,
                    "reasoning_tokens": 2836052,
                    "cache_read_tokens": 36201515,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "alibaba/qwen3.8-27b": {
                "accuracy": 39.056,
                "latency": 5924.968,
                "stderr": 4.938,
                "cost_per_test": 3.553383,
                "token_totals": {
                    "input_tokens": 100712761,
                    "output_tokens": 3731447,
                    "reasoning_tokens": 3412371,
                    "cache_read_tokens": 90857813,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "stepfun/step-5-preview": {
                "accuracy": 38.056,
                "latency": 4820.902,
                "stderr": 6.69,
                "cost_per_test": 2.025817,
                "token_totals": {
                    "input_tokens": 87277108,
                    "output_tokens": 1674540,
                    "reasoning_tokens": 0,
                    "cache_read_tokens": 84779520,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 1024000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Stepfun",
                "harness": null
            },
            "google/gemini-3.6-flash": {
                "accuracy": 35.056,
                "latency": 2903.307,
                "stderr": 5.695,
                "cost_per_test": 8.66611,
                "token_totals": {
                    "input_tokens": 238444814,
                    "output_tokens": 1016307,
                    "reasoning_tokens": 489296,
                    "cache_read_tokens": 232068785,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "deepseek/deepseek-v4-flash-0731": {
                "accuracy": 32.722,
                "latency": 1663.091,
                "stderr": 2.35,
                "cost_per_test": 0.373112,
                "token_totals": {
                    "input_tokens": 53634013,
                    "output_tokens": 1034888,
                    "reasoning_tokens": 889275,
                    "cache_read_tokens": 53385216,
                    "cache_write_tokens": 0
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "meta/muse_spark_1_2": {
                "accuracy": 21.778,
                "latency": 1534.626,
                "stderr": 0.873,
                "cost_per_test": 2.640493,
                "token_totals": {
                    "input_tokens": 62253052,
                    "output_tokens": 1065782,
                    "reasoning_tokens": 811221,
                    "cache_read_tokens": 60536406,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "thinkingmachines/inkling": {
                "accuracy": 14.944,
                "latency": 1812.242,
                "stderr": 5.403,
                "cost_per_test": 1.70838,
                "token_totals": {
                    "input_tokens": 40483119,
                    "output_tokens": 657236,
                    "reasoning_tokens": 0,
                    "cache_read_tokens": 39662208,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "thinkingmachines/inkling-small": {
                "accuracy": 9.333,
                "latency": 678.056,
                "stderr": 4.475,
                "cost_per_test": 0.337871,
                "token_totals": {
                    "input_tokens": 20758234,
                    "output_tokens": 519427,
                    "reasoning_tokens": 0,
                    "cache_read_tokens": 20156416,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 262000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "inception/mercury-2.5": {
                "accuracy": 2.444,
                "latency": 242.94,
                "stderr": 0.683,
                "cost_per_test": 0.160874,
                "token_totals": {
                    "input_tokens": 11229915,
                    "output_tokens": 201476,
                    "reasoning_tokens": 194094,
                    "cache_read_tokens": 8024752,
                    "cache_write_tokens": 0
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Inception",
                "harness": null
            }
        },
        "ioi2024": {
            "google/gemini-4-argon": {
                "accuracy": 100.0,
                "latency": 702.579,
                "stderr": 0.0,
                "cost_per_test": 2.322254,
                "token_totals": {
                    "input_tokens": 7390720,
                    "output_tokens": 353414,
                    "reasoning_tokens": 269441,
                    "cache_read_tokens": 5973062,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-5-5": {
                "accuracy": 100.0,
                "latency": 853.121,
                "stderr": 0.0,
                "cost_per_test": 3.663947,
                "token_totals": {
                    "input_tokens": 19434089,
                    "output_tokens": 474742,
                    "reasoning_tokens": 383878,
                    "cache_read_tokens": 18408980,
                    "cache_write_tokens": 1024815
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-fable-5-1": {
                "accuracy": 100.0,
                "latency": 989.274,
                "stderr": 0.0,
                "cost_per_test": 8.136653,
                "token_totals": {
                    "input_tokens": 24495791,
                    "output_tokens": 641287,
                    "reasoning_tokens": 543804,
                    "cache_read_tokens": 23627831,
                    "cache_write_tokens": 867604
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-6-astra": {
                "accuracy": 100.0,
                "latency": 1078.132,
                "stderr": 0.0,
                "cost_per_test": 3.359584,
                "token_totals": {
                    "input_tokens": 1364710,
                    "output_tokens": 64652,
                    "reasoning_tokens": 36313,
                    "cache_read_tokens": 1234356,
                    "cache_write_tokens": 130132
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-6.1-sol": {
                "accuracy": 100.0,
                "latency": 1198.518,
                "stderr": 0.0,
                "cost_per_test": 0.780945,
                "token_totals": {
                    "input_tokens": 1641035,
                    "output_tokens": 91097,
                    "reasoning_tokens": 62562,
                    "cache_read_tokens": 1475302,
                    "cache_write_tokens": 165481
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-6-sol": {
                "accuracy": 100.0,
                "latency": 1861.244,
                "stderr": 0.0,
                "cost_per_test": 2.040245,
                "token_totals": {
                    "input_tokens": 3834387,
                    "output_tokens": 155914,
                    "reasoning_tokens": 120498,
                    "cache_read_tokens": 3534365,
                    "cache_write_tokens": 299671
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-5": {
                "accuracy": 100.0,
                "latency": 2484.164,
                "stderr": 0.0,
                "cost_per_test": 9.345056,
                "token_totals": {
                    "input_tokens": 48319012,
                    "output_tokens": 943409,
                    "reasoning_tokens": 753737,
                    "cache_read_tokens": 46870966,
                    "cache_write_tokens": 1447518
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.6-sol": {
                "accuracy": 100.0,
                "latency": 2893.045,
                "stderr": 0.0,
                "cost_per_test": 6.687056,
                "token_totals": {
                    "input_tokens": 3559180,
                    "output_tokens": 137780,
                    "reasoning_tokens": 95448,
                    "cache_read_tokens": 3331853,
                    "cache_write_tokens": 226973
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-sonnet-5-5": {
                "accuracy": 85.0,
                "latency": 2893.871,
                "stderr": 13.693,
                "cost_per_test": 5.767053,
                "token_totals": {
                    "input_tokens": 55411541,
                    "output_tokens": 1216766,
                    "reasoning_tokens": 1031007,
                    "cache_read_tokens": 50475623,
                    "cache_write_tokens": 4935394
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.6-luna": {
                "accuracy": 85.0,
                "latency": 4514.485,
                "stderr": 13.693,
                "cost_per_test": 0.68829,
                "token_totals": {
                    "input_tokens": 27327154,
                    "output_tokens": 405689,
                    "reasoning_tokens": 306881,
                    "cache_read_tokens": 26448976,
                    "cache_write_tokens": 877395
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.6-terra": {
                "accuracy": 85.0,
                "latency": 5068.741,
                "stderr": 13.693,
                "cost_per_test": 8.916335,
                "token_totals": {
                    "input_tokens": 12586377,
                    "output_tokens": 239591,
                    "reasoning_tokens": 185477,
                    "cache_read_tokens": 12173912,
                    "cache_write_tokens": 411829
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3.8-max": {
                "accuracy": 78.167,
                "latency": 5942.148,
                "stderr": 13.865,
                "cost_per_test": 8.225132,
                "token_totals": {
                    "input_tokens": 141837625,
                    "output_tokens": 1432357,
                    "reasoning_tokens": 1171711,
                    "cache_read_tokens": 139725952,
                    "cache_write_tokens": 0
                },
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "zai/glm-5.3": {
                "accuracy": 76.333,
                "latency": 4368.615,
                "stderr": 14.379,
                "cost_per_test": 4.958568,
                "token_totals": {
                    "input_tokens": 77291493,
                    "output_tokens": 1413690,
                    "reasoning_tokens": 1296076,
                    "cache_read_tokens": 75056960,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "google/gemini-3.7-flash": {
                "accuracy": 72.833,
                "latency": 549.766,
                "stderr": 15.812,
                "cost_per_test": 2.498747,
                "token_totals": {
                    "input_tokens": 33091114,
                    "output_tokens": 605650,
                    "reasoning_tokens": 336489,
                    "cache_read_tokens": 29027083,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-6-luna": {
                "accuracy": 69.333,
                "latency": 2908.975,
                "stderr": 14.061,
                "cost_per_test": 0.280681,
                "token_totals": {
                    "input_tokens": 33390628,
                    "output_tokens": 703757,
                    "reasoning_tokens": 639764,
                    "cache_read_tokens": 32253025,
                    "cache_write_tokens": 1137051
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "fireworks/ember-1": {
                "accuracy": 64.5,
                "latency": 3399.993,
                "stderr": 16.162,
                "cost_per_test": 10.548392,
                "token_totals": {
                    "input_tokens": 108709099,
                    "output_tokens": 1278995,
                    "reasoning_tokens": 1147844,
                    "cache_read_tokens": 104476461,
                    "cache_write_tokens": 0
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "openai/gpt-5.3-codex": {
                "accuracy": 62.833,
                "latency": 2616.979,
                "stderr": 17.092,
                "cost_per_test": 3.109282,
                "token_totals": {
                    "input_tokens": 16836055,
                    "output_tokens": 411919,
                    "reasoning_tokens": 325681,
                    "cache_read_tokens": 16113664,
                    "cache_write_tokens": 0
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4.7": {
                "accuracy": 61.667,
                "latency": 2486.576,
                "stderr": 15.886,
                "cost_per_test": 8.910759,
                "token_totals": {
                    "input_tokens": 48099946,
                    "output_tokens": 1068040,
                    "reasoning_tokens": 853673,
                    "cache_read_tokens": 41838592,
                    "cache_write_tokens": 0
                },
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "kimi/kimi-k3": {
                "accuracy": 59.667,
                "latency": 10134.006,
                "stderr": 16.636,
                "cost_per_test": 7.315656,
                "token_totals": {
                    "input_tokens": 57291796,
                    "output_tokens": 1284464,
                    "reasoning_tokens": 1102800,
                    "cache_read_tokens": 54536448,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "meta/muse_spark_1_3_max": {
                "accuracy": 56.0,
                "latency": 1765.929,
                "stderr": 15.697,
                "cost_per_test": 1.463371,
                "token_totals": {
                    "input_tokens": 18194409,
                    "output_tokens": 1092665,
                    "reasoning_tokens": 966676,
                    "cache_read_tokens": 16977503,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "google/gemini-3.8-flash": {
                "accuracy": 55.667,
                "latency": 938.738,
                "stderr": 14.617,
                "cost_per_test": 4.196808,
                "token_totals": {
                    "input_tokens": 89097624,
                    "output_tokens": 890798,
                    "reasoning_tokens": 523067,
                    "cache_read_tokens": 85293758,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 55.5,
                "latency": 585.596,
                "stderr": 18.714,
                "cost_per_test": 1.207234,
                "token_totals": {
                    "input_tokens": 7142113,
                    "output_tokens": 306623,
                    "reasoning_tokens": 243010,
                    "cache_read_tokens": 5955720,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "tencent/hy4-preview": {
                "accuracy": 54.167,
                "latency": 3412.169,
                "stderr": 16.922,
                "cost_per_test": 1.261586,
                "token_totals": {
                    "input_tokens": 54985408,
                    "output_tokens": 1133101,
                    "reasoning_tokens": 1008687,
                    "cache_read_tokens": 52125312,
                    "cache_write_tokens": 0
                },
                "temperature": 0.9,
                "top_p": 1.0,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Tencent",
                "harness": null
            },
            "zai/glm-5.3-flash": {
                "accuracy": 53.167,
                "latency": 6676.046,
                "stderr": 15.536,
                "cost_per_test": 0.645447,
                "token_totals": {
                    "input_tokens": 173274099,
                    "output_tokens": 1305553,
                    "reasoning_tokens": 1174181,
                    "cache_read_tokens": 164926080,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "anthropic/claude-sonnet-5": {
                "accuracy": 50.5,
                "latency": 3853.249,
                "stderr": 13.751,
                "cost_per_test": 12.245865,
                "token_totals": {
                    "input_tokens": 205900188,
                    "output_tokens": 2113886,
                    "reasoning_tokens": 1820456,
                    "cache_read_tokens": 201049345,
                    "cache_write_tokens": 4849553
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "xiaomi/mimo-v2.6-flash": {
                "accuracy": 50.5,
                "latency": 6891.616,
                "stderr": 12.142,
                "cost_per_test": 0.247075,
                "token_totals": {
                    "input_tokens": 147537013,
                    "output_tokens": 1725094,
                    "reasoning_tokens": 1533007,
                    "cache_read_tokens": 144385600,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "grok/grok-4.6": {
                "accuracy": 50.333,
                "latency": 14058.254,
                "stderr": 16.482,
                "cost_per_test": 7.380562,
                "token_totals": {
                    "input_tokens": 21028515,
                    "output_tokens": 4867447,
                    "reasoning_tokens": 4523512,
                    "cache_read_tokens": 18032896,
                    "cache_write_tokens": 0
                },
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "mistralai/mistral-large-4": {
                "accuracy": 50.167,
                "latency": 8012.36,
                "stderr": 16.522,
                "cost_per_test": 8.854199,
                "token_totals": {
                    "input_tokens": 117325391,
                    "output_tokens": 4925557,
                    "reasoning_tokens": 0,
                    "cache_read_tokens": 104119808,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "deepseek/deepseek-v4-pro-0813": {
                "accuracy": 50.0,
                "latency": 3096.029,
                "stderr": 16.444,
                "cost_per_test": 1.974595,
                "token_totals": {
                    "input_tokens": 118643031,
                    "output_tokens": 1592017,
                    "reasoning_tokens": 1385043,
                    "cache_read_tokens": 118412672,
                    "cache_write_tokens": 0
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "grok/grok-4.5": {
                "accuracy": 47.167,
                "latency": 6236.578,
                "stderr": 11.845,
                "cost_per_test": 4.351491,
                "token_totals": {
                    "input_tokens": 20938338,
                    "output_tokens": 1900714,
                    "reasoning_tokens": 1375170,
                    "cache_read_tokens": 16010624,
                    "cache_write_tokens": 0
                },
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "xiaomi/mimo-v2.6-pro": {
                "accuracy": 43.5,
                "latency": 7363.258,
                "stderr": 17.084,
                "cost_per_test": 0.787332,
                "token_totals": {
                    "input_tokens": 26943589,
                    "output_tokens": 3050999,
                    "reasoning_tokens": 3019947,
                    "cache_read_tokens": 23086848,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "deepseek/deepseek-v4.1-flash": {
                "accuracy": 42.667,
                "latency": 914.56,
                "stderr": 14.693,
                "cost_per_test": 0.207781,
                "token_totals": {
                    "input_tokens": 28008480,
                    "output_tokens": 856811,
                    "reasoning_tokens": 758852,
                    "cache_read_tokens": 27861632,
                    "cache_write_tokens": 0
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "meta/muse_spark_1_3": {
                "accuracy": 42.667,
                "latency": 1781.576,
                "stderr": 14.985,
                "cost_per_test": 3.031404,
                "token_totals": {
                    "input_tokens": 43374206,
                    "output_tokens": 2335817,
                    "reasoning_tokens": 2233638,
                    "cache_read_tokens": 41843122,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "google/gemini-3.6-flash": {
                "accuracy": 40.667,
                "latency": 1025.59,
                "stderr": 17.597,
                "cost_per_test": 4.655748,
                "token_totals": {
                    "input_tokens": 116644938,
                    "output_tokens": 640960,
                    "reasoning_tokens": 416752,
                    "cache_read_tokens": 112474165,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "deepseek/deepseek-v4-flash-0731": {
                "accuracy": 34.0,
                "latency": 1373.252,
                "stderr": 14.887,
                "cost_per_test": 0.346944,
                "token_totals": {
                    "input_tokens": 56595542,
                    "output_tokens": 905221,
                    "reasoning_tokens": 766588,
                    "cache_read_tokens": 56395904,
                    "cache_write_tokens": 0
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "alibaba/qwen3.8-27b": {
                "accuracy": 31.833,
                "latency": 3824.509,
                "stderr": 15.28,
                "cost_per_test": 2.122262,
                "token_totals": {
                    "input_tokens": 81058514,
                    "output_tokens": 2091540,
                    "reasoning_tokens": 1833773,
                    "cache_read_tokens": 76189696,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "stepfun/step-5-preview": {
                "accuracy": 26.667,
                "latency": 5343.371,
                "stderr": 11.394,
                "cost_per_test": 2.142166,
                "token_totals": {
                    "input_tokens": 82671675,
                    "output_tokens": 2027988,
                    "reasoning_tokens": 0,
                    "cache_read_tokens": 79725056,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 1024000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Stepfun",
                "harness": null
            },
            "meta/muse_spark_1_2": {
                "accuracy": 22.0,
                "latency": 1236.283,
                "stderr": 12.131,
                "cost_per_test": 2.353067,
                "token_totals": {
                    "input_tokens": 56226636,
                    "output_tokens": 966428,
                    "reasoning_tokens": 743553,
                    "cache_read_tokens": 54859363,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "thinkingmachines/inkling-small": {
                "accuracy": 17.167,
                "latency": 426.534,
                "stderr": 15.13,
                "cost_per_test": 0.220122,
                "token_totals": {
                    "input_tokens": 10894488,
                    "output_tokens": 449267,
                    "reasoning_tokens": 0,
                    "cache_read_tokens": 10402944,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 262000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "thinkingmachines/inkling": {
                "accuracy": 4.167,
                "latency": 1907.44,
                "stderr": 3.286,
                "cost_per_test": 2.268219,
                "token_totals": {
                    "input_tokens": 57076005,
                    "output_tokens": 764042,
                    "reasoning_tokens": 0,
                    "cache_read_tokens": 56126720,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "inception/mercury-2.5": {
                "accuracy": 2.667,
                "latency": 203.195,
                "stderr": 1.61,
                "cost_per_test": 0.132547,
                "token_totals": {
                    "input_tokens": 9837909,
                    "output_tokens": 182920,
                    "reasoning_tokens": 175725,
                    "cache_read_tokens": 7344903,
                    "cache_write_tokens": 0
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Inception",
                "harness": null
            }
        },
        "ioi2025": {
            "anthropic/claude-opus-5-5": {
                "accuracy": 100.0,
                "latency": 441.464,
                "stderr": 0.0,
                "cost_per_test": 3.061283,
                "token_totals": {
                    "input_tokens": 13639029,
                    "output_tokens": 497439,
                    "reasoning_tokens": 409486,
                    "cache_read_tokens": 12994777,
                    "cache_write_tokens": 644000
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-4-argon": {
                "accuracy": 100.0,
                "latency": 1034.504,
                "stderr": 0.0,
                "cost_per_test": 2.880431,
                "token_totals": {
                    "input_tokens": 18679947,
                    "output_tokens": 450877,
                    "reasoning_tokens": 288219,
                    "cache_read_tokens": 17488090,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-6.1-sol": {
                "accuracy": 100.0,
                "latency": 1321.855,
                "stderr": 0.0,
                "cost_per_test": 0.813679,
                "token_totals": {
                    "input_tokens": 2020492,
                    "output_tokens": 79067,
                    "reasoning_tokens": 45902,
                    "cache_read_tokens": 1808231,
                    "cache_write_tokens": 212015
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-6-astra": {
                "accuracy": 100.0,
                "latency": 1562.492,
                "stderr": 0.0,
                "cost_per_test": 5.113398,
                "token_totals": {
                    "input_tokens": 2091102,
                    "output_tokens": 70089,
                    "reasoning_tokens": 38414,
                    "cache_read_tokens": 1910090,
                    "cache_write_tokens": 180790
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-sonnet-5-5": {
                "accuracy": 88.333,
                "latency": 1746.178,
                "stderr": 10.65,
                "cost_per_test": 4.613218,
                "token_totals": {
                    "input_tokens": 50307537,
                    "output_tokens": 1282221,
                    "reasoning_tokens": 1122840,
                    "cache_read_tokens": 48266419,
                    "cache_write_tokens": 2040622
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.6-sol": {
                "accuracy": 88.333,
                "latency": 3965.785,
                "stderr": 10.65,
                "cost_per_test": 8.93332,
                "token_totals": {
                    "input_tokens": 5615082,
                    "output_tokens": 205890,
                    "reasoning_tokens": 160719,
                    "cache_read_tokens": 4977405,
                    "cache_write_tokens": 637272
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-fable-5-1": {
                "accuracy": 87.167,
                "latency": 2233.026,
                "stderr": 11.715,
                "cost_per_test": 12.516821,
                "token_totals": {
                    "input_tokens": 40988406,
                    "output_tokens": 1183945,
                    "reasoning_tokens": 1034144,
                    "cache_read_tokens": 39288242,
                    "cache_write_tokens": 1699728
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-5": {
                "accuracy": 87.167,
                "latency": 3406.434,
                "stderr": 11.715,
                "cost_per_test": 16.518293,
                "token_totals": {
                    "input_tokens": 97589394,
                    "output_tokens": 1411316,
                    "reasoning_tokens": 1151620,
                    "cache_read_tokens": 94974943,
                    "cache_write_tokens": 2613707
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-6-sol": {
                "accuracy": 79.333,
                "latency": 1975.572,
                "stderr": 11.934,
                "cost_per_test": 3.361427,
                "token_totals": {
                    "input_tokens": 4500601,
                    "output_tokens": 127414,
                    "reasoning_tokens": 98238,
                    "cache_read_tokens": 4091306,
                    "cache_write_tokens": 408977
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.6-terra": {
                "accuracy": 77.833,
                "latency": 4635.842,
                "stderr": 12.824,
                "cost_per_test": 9.87522,
                "token_totals": {
                    "input_tokens": 8456064,
                    "output_tokens": 262603,
                    "reasoning_tokens": 224508,
                    "cache_read_tokens": 7976676,
                    "cache_write_tokens": 478896
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-5.3": {
                "accuracy": 75.5,
                "latency": 9972.585,
                "stderr": 14.169,
                "cost_per_test": 12.128308,
                "token_totals": {
                    "input_tokens": 194994471,
                    "output_tokens": 2691787,
                    "reasoning_tokens": 2526139,
                    "cache_read_tokens": 186096192,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "google/gemini-3.7-flash": {
                "accuracy": 69.833,
                "latency": 781.833,
                "stderr": 15.185,
                "cost_per_test": 2.302837,
                "token_totals": {
                    "input_tokens": 33069022,
                    "output_tokens": 586720,
                    "reasoning_tokens": 334266,
                    "cache_read_tokens": 29768080,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "tencent/hy4-preview": {
                "accuracy": 68.5,
                "latency": 4036.888,
                "stderr": 12.965,
                "cost_per_test": 1.397613,
                "token_totals": {
                    "input_tokens": 68874783,
                    "output_tokens": 1304192,
                    "reasoning_tokens": 1165941,
                    "cache_read_tokens": 66479168,
                    "cache_write_tokens": 0
                },
                "temperature": 0.9,
                "top_p": 1.0,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Tencent",
                "harness": null
            },
            "alibaba/qwen3.8-max": {
                "accuracy": 68.333,
                "latency": 9212.428,
                "stderr": 18.321,
                "cost_per_test": 15.58529,
                "token_totals": {
                    "input_tokens": 290715071,
                    "output_tokens": 2177084,
                    "reasoning_tokens": 1903731,
                    "cache_read_tokens": 287722368,
                    "cache_write_tokens": 0
                },
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "google/gemini-3.8-flash": {
                "accuracy": 61.333,
                "latency": 697.499,
                "stderr": 15.01,
                "cost_per_test": 3.559997,
                "token_totals": {
                    "input_tokens": 71610674,
                    "output_tokens": 808776,
                    "reasoning_tokens": 576185,
                    "cache_read_tokens": 68238407,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "meta/muse_spark_1_3_max": {
                "accuracy": 61.167,
                "latency": 1740.022,
                "stderr": 13.046,
                "cost_per_test": 2.283476,
                "token_totals": {
                    "input_tokens": 34708381,
                    "output_tokens": 1510730,
                    "reasoning_tokens": 1352239,
                    "cache_read_tokens": 32874926,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "openai/gpt-5.3-codex": {
                "accuracy": 57.0,
                "latency": 1321.093,
                "stderr": 15.452,
                "cost_per_test": 2.174323,
                "token_totals": {
                    "input_tokens": 18182442,
                    "output_tokens": 592046,
                    "reasoning_tokens": 451657,
                    "cache_read_tokens": 17182208,
                    "cache_write_tokens": 0
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-5.3-flash": {
                "accuracy": 56.667,
                "latency": 5866.516,
                "stderr": 14.892,
                "cost_per_test": 0.395586,
                "token_totals": {
                    "input_tokens": 95698762,
                    "output_tokens": 1784337,
                    "reasoning_tokens": 1661062,
                    "cache_read_tokens": 89353088,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "grok/grok-4.7": {
                "accuracy": 56.167,
                "latency": 3028.441,
                "stderr": 15.587,
                "cost_per_test": 13.714707,
                "token_totals": {
                    "input_tokens": 72503645,
                    "output_tokens": 1326776,
                    "reasoning_tokens": 1039852,
                    "cache_read_tokens": 67684352,
                    "cache_write_tokens": 0
                },
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "xiaomi/mimo-v2.6-flash": {
                "accuracy": 54.167,
                "latency": 7184.077,
                "stderr": 12.553,
                "cost_per_test": 0.256024,
                "token_totals": {
                    "input_tokens": 166224962,
                    "output_tokens": 1477004,
                    "reasoning_tokens": 1272448,
                    "cache_read_tokens": 163619392,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "openai/gpt-5.6-luna": {
                "accuracy": 49.833,
                "latency": 4248.955,
                "stderr": 15.446,
                "cost_per_test": 0.543819,
                "token_totals": {
                    "input_tokens": 55137587,
                    "output_tokens": 489046,
                    "reasoning_tokens": 356631,
                    "cache_read_tokens": 54119432,
                    "cache_write_tokens": 1016865
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "stepfun/step-5-preview": {
                "accuracy": 49.833,
                "latency": 5117.861,
                "stderr": 17.718,
                "cost_per_test": 2.286536,
                "token_totals": {
                    "input_tokens": 109078778,
                    "output_tokens": 1669504,
                    "reasoning_tokens": 0,
                    "cache_read_tokens": 106455808,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 1024000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Stepfun",
                "harness": null
            },
            "grok/grok-4.6": {
                "accuracy": 49.0,
                "latency": 19383.659,
                "stderr": 15.224,
                "cost_per_test": 12.803831,
                "token_totals": {
                    "input_tokens": 55759220,
                    "output_tokens": 6932737,
                    "reasoning_tokens": 6403804,
                    "cache_read_tokens": 50955520,
                    "cache_write_tokens": 0
                },
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "deepseek/deepseek-v4-pro-0813": {
                "accuracy": 48.5,
                "latency": 6167.407,
                "stderr": 14.041,
                "cost_per_test": 2.620016,
                "token_totals": {
                    "input_tokens": 161838645,
                    "output_tokens": 2026899,
                    "reasoning_tokens": 1719775,
                    "cache_read_tokens": 161448832,
                    "cache_write_tokens": 0
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "alibaba/qwen3.8-27b": {
                "accuracy": 48.5,
                "latency": 10294.149,
                "stderr": 15.556,
                "cost_per_test": 6.569652,
                "token_totals": {
                    "input_tokens": 148411588,
                    "output_tokens": 7232117,
                    "reasoning_tokens": 6755518,
                    "cache_read_tokens": 128495616,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 46.833,
                "latency": 917.103,
                "stderr": 13.181,
                "cost_per_test": 1.414465,
                "token_totals": {
                    "input_tokens": 7930164,
                    "output_tokens": 361796,
                    "reasoning_tokens": 298868,
                    "cache_read_tokens": 6508383,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "meta/muse_spark_1_3": {
                "accuracy": 46.833,
                "latency": 2243.31,
                "stderr": 13.386,
                "cost_per_test": 2.451039,
                "token_totals": {
                    "input_tokens": 24665482,
                    "output_tokens": 2228922,
                    "reasoning_tokens": 2115096,
                    "cache_read_tokens": 23369084,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "anthropic/claude-sonnet-5": {
                "accuracy": 42.0,
                "latency": 4427.238,
                "stderr": 15.19,
                "cost_per_test": 14.811094,
                "token_totals": {
                    "input_tokens": 252035458,
                    "output_tokens": 2410105,
                    "reasoning_tokens": 2121726,
                    "cache_read_tokens": 245792345,
                    "cache_write_tokens": 6241633
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.6-flash": {
                "accuracy": 40.833,
                "latency": 935.617,
                "stderr": 9.711,
                "cost_per_test": 1.747464,
                "token_totals": {
                    "input_tokens": 22669003,
                    "output_tokens": 527617,
                    "reasoning_tokens": 382252,
                    "cache_read_tokens": 20352477,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "xiaomi/mimo-v2.6-pro": {
                "accuracy": 39.333,
                "latency": 13526.414,
                "stderr": 18.072,
                "cost_per_test": 1.154221,
                "token_totals": {
                    "input_tokens": 65737574,
                    "output_tokens": 3554858,
                    "reasoning_tokens": 3491200,
                    "cache_read_tokens": 61284224,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "openai/gpt-6-luna": {
                "accuracy": 39.0,
                "latency": 1692.641,
                "stderr": 17.408,
                "cost_per_test": 0.155148,
                "token_totals": {
                    "input_tokens": 18162547,
                    "output_tokens": 604036,
                    "reasoning_tokens": 548590,
                    "cache_read_tokens": 17298254,
                    "cache_write_tokens": 863717
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "mistralai/mistral-large-4": {
                "accuracy": 37.833,
                "latency": 5151.666,
                "stderr": 16.767,
                "cost_per_test": 4.58407,
                "token_totals": {
                    "input_tokens": 51341971,
                    "output_tokens": 2657398,
                    "reasoning_tokens": 0,
                    "cache_read_tokens": 43793920,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "deepseek/deepseek-v4.1-flash": {
                "accuracy": 35.167,
                "latency": 1631.199,
                "stderr": 9.182,
                "cost_per_test": 0.297858,
                "token_totals": {
                    "input_tokens": 51333215,
                    "output_tokens": 1169612,
                    "reasoning_tokens": 1027911,
                    "cache_read_tokens": 51142912,
                    "cache_write_tokens": 0
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "fireworks/ember-1": {
                "accuracy": 35.167,
                "latency": 6604.095,
                "stderr": 10.789,
                "cost_per_test": 20.898492,
                "token_totals": {
                    "input_tokens": 254434959,
                    "output_tokens": 2076000,
                    "reasoning_tokens": 1752753,
                    "cache_read_tokens": 247865539,
                    "cache_write_tokens": 0
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "grok/grok-4.5": {
                "accuracy": 33.5,
                "latency": 5571.97,
                "stderr": 10.941,
                "cost_per_test": 9.819693,
                "token_totals": {
                    "input_tokens": 75163876,
                    "output_tokens": 1611916,
                    "reasoning_tokens": 849073,
                    "cache_read_tokens": 69449344,
                    "cache_write_tokens": 0
                },
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "kimi/kimi-k3": {
                "accuracy": 29.333,
                "latency": 20340.198,
                "stderr": 15.07,
                "cost_per_test": 22.111088,
                "token_totals": {
                    "input_tokens": 228599372,
                    "output_tokens": 2631940,
                    "reasoning_tokens": 2379038,
                    "cache_read_tokens": 219485440,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "deepseek/deepseek-v4-flash-0731": {
                "accuracy": 28.167,
                "latency": 1580.139,
                "stderr": 10.326,
                "cost_per_test": 0.381691,
                "token_totals": {
                    "input_tokens": 49247535,
                    "output_tokens": 1134860,
                    "reasoning_tokens": 981192,
                    "cache_read_tokens": 49070848,
                    "cache_write_tokens": 0
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "meta/muse_spark_1_2": {
                "accuracy": 23.167,
                "latency": 2103.366,
                "stderr": 10.34,
                "cost_per_test": 3.879691,
                "token_totals": {
                    "input_tokens": 97687063,
                    "output_tokens": 1350840,
                    "reasoning_tokens": 1026584,
                    "cache_read_tokens": 95167310,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "thinkingmachines/inkling": {
                "accuracy": 21.0,
                "latency": 2051.208,
                "stderr": 10.883,
                "cost_per_test": 1.700535,
                "token_totals": {
                    "input_tokens": 41444520,
                    "output_tokens": 607850,
                    "reasoning_tokens": 0,
                    "cache_read_tokens": 40637568,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "thinkingmachines/inkling-small": {
                "accuracy": 9.167,
                "latency": 540.841,
                "stderr": 6.568,
                "cost_per_test": 0.297854,
                "token_totals": {
                    "input_tokens": 17590551,
                    "output_tokens": 476309,
                    "reasoning_tokens": 0,
                    "cache_read_tokens": 17016832,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 262000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "inception/mercury-2.5": {
                "accuracy": 3.5,
                "latency": 239.919,
                "stderr": 2.685,
                "cost_per_test": 0.187303,
                "token_totals": {
                    "input_tokens": 11283658,
                    "output_tokens": 173264,
                    "reasoning_tokens": 165909,
                    "cache_read_tokens": 7109497,
                    "cache_write_tokens": 0
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Inception",
                "harness": null
            }
        },
        "ioi2026": {
            "google/gemini-4-argon": {
                "accuracy": 100.0,
                "latency": 2640.726,
                "stderr": 0.0,
                "cost_per_test": 10.720193,
                "token_totals": {
                    "input_tokens": 119965969,
                    "output_tokens": 1055816,
                    "reasoning_tokens": 778031,
                    "cache_read_tokens": 114910273,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-6-astra": {
                "accuracy": 100.0,
                "latency": 3058.726,
                "stderr": 0.0,
                "cost_per_test": 11.012171,
                "token_totals": {
                    "input_tokens": 1815327,
                    "output_tokens": 78886,
                    "reasoning_tokens": 47036,
                    "cache_read_tokens": 1626094,
                    "cache_write_tokens": 188984
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.6-terra": {
                "accuracy": 100.0,
                "latency": 5273.93,
                "stderr": 0.0,
                "cost_per_test": 7.20047,
                "token_totals": {
                    "input_tokens": 4324713,
                    "output_tokens": 165665,
                    "reasoning_tokens": 136111,
                    "cache_read_tokens": 3982511,
                    "cache_write_tokens": 341800
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-6.1-sol": {
                "accuracy": 90.667,
                "latency": 3151.77,
                "stderr": 8.52,
                "cost_per_test": 2.190169,
                "token_totals": {
                    "input_tokens": 2088802,
                    "output_tokens": 104447,
                    "reasoning_tokens": 76626,
                    "cache_read_tokens": 1851053,
                    "cache_write_tokens": 237506
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-5-5": {
                "accuracy": 85.167,
                "latency": 1810.988,
                "stderr": 13.541,
                "cost_per_test": 9.050024,
                "token_totals": {
                    "input_tokens": 49340144,
                    "output_tokens": 1260558,
                    "reasoning_tokens": 1097168,
                    "cache_read_tokens": 47448899,
                    "cache_write_tokens": 1890851
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-fable-5-1": {
                "accuracy": 85.167,
                "latency": 2195.415,
                "stderr": 13.541,
                "cost_per_test": 12.937402,
                "token_totals": {
                    "input_tokens": 44567310,
                    "output_tokens": 1121365,
                    "reasoning_tokens": 983228,
                    "cache_read_tokens": 42784481,
                    "cache_write_tokens": 1782351
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.6-sol": {
                "accuracy": 85.167,
                "latency": 4276.062,
                "stderr": 13.541,
                "cost_per_test": 8.32603,
                "token_totals": {
                    "input_tokens": 4294481,
                    "output_tokens": 125409,
                    "reasoning_tokens": 87619,
                    "cache_read_tokens": 4003687,
                    "cache_write_tokens": 290401
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-sonnet-5-5": {
                "accuracy": 75.833,
                "latency": 3613.203,
                "stderr": 14.485,
                "cost_per_test": 12.192486,
                "token_totals": {
                    "input_tokens": 214699529,
                    "output_tokens": 2037311,
                    "reasoning_tokens": 1712396,
                    "cache_read_tokens": 210420244,
                    "cache_write_tokens": 4278377
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-6-sol": {
                "accuracy": 68.5,
                "latency": 1675.844,
                "stderr": 18.233,
                "cost_per_test": 2.708373,
                "token_totals": {
                    "input_tokens": 3849795,
                    "output_tokens": 158823,
                    "reasoning_tokens": 125568,
                    "cache_read_tokens": 3583111,
                    "cache_write_tokens": 266327
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-5": {
                "accuracy": 65.833,
                "latency": 4315.537,
                "stderr": 14.58,
                "cost_per_test": 23.568374,
                "token_totals": {
                    "input_tokens": 142324076,
                    "output_tokens": 1803821,
                    "reasoning_tokens": 1472640,
                    "cache_read_tokens": 137949530,
                    "cache_write_tokens": 4373778
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.7-flash": {
                "accuracy": 60.833,
                "latency": 2475.465,
                "stderr": 16.627,
                "cost_per_test": 5.392624,
                "token_totals": {
                    "input_tokens": 70967304,
                    "output_tokens": 967801,
                    "reasoning_tokens": 429792,
                    "cache_read_tokens": 60262017,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3.8-max": {
                "accuracy": 60.167,
                "latency": 4040.063,
                "stderr": 16.75,
                "cost_per_test": 4.596134,
                "token_totals": {
                    "input_tokens": 68265157,
                    "output_tokens": 1024330,
                    "reasoning_tokens": 884682,
                    "cache_read_tokens": 66689408,
                    "cache_write_tokens": 0
                },
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/gpt-6-luna": {
                "accuracy": 58.333,
                "latency": 1839.807,
                "stderr": 17.704,
                "cost_per_test": 0.162408,
                "token_totals": {
                    "input_tokens": 18875849,
                    "output_tokens": 785789,
                    "reasoning_tokens": 740012,
                    "cache_read_tokens": 17849572,
                    "cache_write_tokens": 1025842
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "kimi/kimi-k3": {
                "accuracy": 57.833,
                "latency": 16051.546,
                "stderr": 17.517,
                "cost_per_test": 13.068357,
                "token_totals": {
                    "input_tokens": 100064871,
                    "output_tokens": 2225538,
                    "reasoning_tokens": 1908641,
                    "cache_read_tokens": 94506496,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "deepseek/deepseek-v4-pro-0813": {
                "accuracy": 56.333,
                "latency": 2874.481,
                "stderr": 18.038,
                "cost_per_test": 1.860156,
                "token_totals": {
                    "input_tokens": 100059606,
                    "output_tokens": 1594019,
                    "reasoning_tokens": 1391854,
                    "cache_read_tokens": 99732352,
                    "cache_write_tokens": 0
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "grok/grok-4.7": {
                "accuracy": 55.333,
                "latency": 2910.428,
                "stderr": 16.711,
                "cost_per_test": 15.498257,
                "token_totals": {
                    "input_tokens": 71444691,
                    "output_tokens": 1126422,
                    "reasoning_tokens": 879581,
                    "cache_read_tokens": 64031104,
                    "cache_write_tokens": 0
                },
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "tencent/hy4-preview": {
                "accuracy": 55.333,
                "latency": 3450.828,
                "stderr": 15.13,
                "cost_per_test": 1.405179,
                "token_totals": {
                    "input_tokens": 82393256,
                    "output_tokens": 1209254,
                    "reasoning_tokens": 1057668,
                    "cache_read_tokens": 80146048,
                    "cache_write_tokens": 0
                },
                "temperature": 0.9,
                "top_p": 1.0,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Tencent",
                "harness": null
            },
            "google/gemini-3.8-flash": {
                "accuracy": 53.833,
                "latency": 945.157,
                "stderr": 14.628,
                "cost_per_test": 4.171352,
                "token_totals": {
                    "input_tokens": 81572346,
                    "output_tokens": 1014339,
                    "reasoning_tokens": 633191,
                    "cache_read_tokens": 77731817,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "zai/glm-5.3": {
                "accuracy": 53.5,
                "latency": 4782.195,
                "stderr": 19.038,
                "cost_per_test": 5.925734,
                "token_totals": {
                    "input_tokens": 87654147,
                    "output_tokens": 1633347,
                    "reasoning_tokens": 1502592,
                    "cache_read_tokens": 82789376,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 53.167,
                "latency": 665.213,
                "stderr": 19.207,
                "cost_per_test": 1.132771,
                "token_totals": {
                    "input_tokens": 6937662,
                    "output_tokens": 295487,
                    "reasoning_tokens": 246940,
                    "cache_read_tokens": 5902523,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "meta/muse_spark_1_3_max": {
                "accuracy": 52.5,
                "latency": 1708.038,
                "stderr": 15.441,
                "cost_per_test": 1.442136,
                "token_totals": {
                    "input_tokens": 21291587,
                    "output_tokens": 995310,
                    "reasoning_tokens": 872131,
                    "cache_read_tokens": 20231774,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "openai/gpt-5.6-luna": {
                "accuracy": 50.5,
                "latency": 3762.795,
                "stderr": 16.768,
                "cost_per_test": 0.506723,
                "token_totals": {
                    "input_tokens": 17518434,
                    "output_tokens": 487964,
                    "reasoning_tokens": 399941,
                    "cache_read_tokens": 16531904,
                    "cache_write_tokens": 985765
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "mistralai/mistral-large-4": {
                "accuracy": 47.833,
                "latency": 1907.262,
                "stderr": 17.392,
                "cost_per_test": 2.739805,
                "token_totals": {
                    "input_tokens": 49111512,
                    "output_tokens": 1080873,
                    "reasoning_tokens": 0,
                    "cache_read_tokens": 44976128,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "zai/glm-5.3-flash": {
                "accuracy": 47.667,
                "latency": 4103.01,
                "stderr": 16.084,
                "cost_per_test": 0.246756,
                "token_totals": {
                    "input_tokens": 47844222,
                    "output_tokens": 1604684,
                    "reasoning_tokens": 1513269,
                    "cache_read_tokens": 43622784,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "grok/grok-4.6": {
                "accuracy": 43.5,
                "latency": 9258.746,
                "stderr": 17.052,
                "cost_per_test": 4.365764,
                "token_totals": {
                    "input_tokens": 7617387,
                    "output_tokens": 3192387,
                    "reasoning_tokens": 2993534,
                    "cache_read_tokens": 5505280,
                    "cache_write_tokens": 0
                },
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "deepseek/deepseek-v4.1-flash": {
                "accuracy": 43.0,
                "latency": 782.974,
                "stderr": 16.931,
                "cost_per_test": 0.222012,
                "token_totals": {
                    "input_tokens": 38321706,
                    "output_tokens": 884107,
                    "reasoning_tokens": 775032,
                    "cache_read_tokens": 38204416,
                    "cache_write_tokens": 0
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "anthropic/claude-sonnet-5": {
                "accuracy": 42.5,
                "latency": 2471.166,
                "stderr": 17.4,
                "cost_per_test": 11.612894,
                "token_totals": {
                    "input_tokens": 224818214,
                    "output_tokens": 1729761,
                    "reasoning_tokens": 1465902,
                    "cache_read_tokens": 221593575,
                    "cache_write_tokens": 3223527
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "meta/muse_spark_1_3": {
                "accuracy": 42.333,
                "latency": 1348.515,
                "stderr": 16.681,
                "cost_per_test": 1.599779,
                "token_totals": {
                    "input_tokens": 17062093,
                    "output_tokens": 1388724,
                    "reasoning_tokens": 1255148,
                    "cache_read_tokens": 16091922,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "openai/gpt-5.3-codex": {
                "accuracy": 41.667,
                "latency": 1436.984,
                "stderr": 12.355,
                "cost_per_test": 2.758885,
                "token_totals": {
                    "input_tokens": 31929158,
                    "output_tokens": 652073,
                    "reasoning_tokens": 484527,
                    "cache_read_tokens": 30763008,
                    "cache_write_tokens": 0
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4.5": {
                "accuracy": 41.0,
                "latency": 3775.907,
                "stderr": 15.936,
                "cost_per_test": 2.742259,
                "token_totals": {
                    "input_tokens": 15615274,
                    "output_tokens": 1043010,
                    "reasoning_tokens": 671701,
                    "cache_read_tokens": 12400896,
                    "cache_write_tokens": 0
                },
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "fireworks/ember-1": {
                "accuracy": 39.333,
                "latency": 9199.271,
                "stderr": 14.554,
                "cost_per_test": 41.928604,
                "token_totals": {
                    "input_tokens": 641612685,
                    "output_tokens": 2134759,
                    "reasoning_tokens": 1673175,
                    "cache_read_tokens": 631612766,
                    "cache_write_tokens": 0
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "xiaomi/mimo-v2.6-flash": {
                "accuracy": 38.333,
                "latency": 5489.124,
                "stderr": 17.908,
                "cost_per_test": 0.142214,
                "token_totals": {
                    "input_tokens": 57211515,
                    "output_tokens": 1064149,
                    "reasoning_tokens": 958931,
                    "cache_read_tokens": 55639616,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "stepfun/step-5-preview": {
                "accuracy": 37.667,
                "latency": 4001.475,
                "stderr": 18.13,
                "cost_per_test": 1.648748,
                "token_totals": {
                    "input_tokens": 70080870,
                    "output_tokens": 1326129,
                    "reasoning_tokens": 0,
                    "cache_read_tokens": 68157696,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 1024000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Stepfun",
                "harness": null
            },
            "alibaba/qwen3.8-27b": {
                "accuracy": 36.833,
                "latency": 3656.248,
                "stderr": 18.319,
                "cost_per_test": 1.968236,
                "token_totals": {
                    "input_tokens": 72668181,
                    "output_tokens": 1870685,
                    "reasoning_tokens": 1647822,
                    "cache_read_tokens": 67888128,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "deepseek/deepseek-v4-flash-0731": {
                "accuracy": 36.0,
                "latency": 2035.883,
                "stderr": 18.529,
                "cost_per_test": 0.3907,
                "token_totals": {
                    "input_tokens": 55058963,
                    "output_tokens": 1064584,
                    "reasoning_tokens": 920044,
                    "cache_read_tokens": 54688896,
                    "cache_write_tokens": 0
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "xiaomi/mimo-v2.6-pro": {
                "accuracy": 35.167,
                "latency": 5457.878,
                "stderr": 18.783,
                "cost_per_test": 0.564938,
                "token_totals": {
                    "input_tokens": 26507616,
                    "output_tokens": 2049508,
                    "reasoning_tokens": 1997010,
                    "cache_read_tokens": 24233472,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "google/gemini-3.6-flash": {
                "accuracy": 23.667,
                "latency": 6748.715,
                "stderr": 14.009,
                "cost_per_test": 19.595118,
                "token_totals": {
                    "input_tokens": 576020501,
                    "output_tokens": 1880343,
                    "reasoning_tokens": 668883,
                    "cache_read_tokens": 563379714,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "meta/muse_spark_1_2": {
                "accuracy": 20.167,
                "latency": 1264.229,
                "stderr": 14.739,
                "cost_per_test": 1.688722,
                "token_totals": {
                    "input_tokens": 32845457,
                    "output_tokens": 880077,
                    "reasoning_tokens": 663527,
                    "cache_read_tokens": 31582544,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "thinkingmachines/inkling": {
                "accuracy": 19.667,
                "latency": 1478.078,
                "stderr": 14.755,
                "cost_per_test": 1.156386,
                "token_totals": {
                    "input_tokens": 22928831,
                    "output_tokens": 599816,
                    "reasoning_tokens": 0,
                    "cache_read_tokens": 22222336,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "thinkingmachines/inkling-small": {
                "accuracy": 1.667,
                "latency": 1066.794,
                "stderr": 0.991,
                "cost_per_test": 0.495638,
                "token_totals": {
                    "input_tokens": 33789663,
                    "output_tokens": 632704,
                    "reasoning_tokens": 0,
                    "cache_read_tokens": 33049472,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 262000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "inception/mercury-2.5": {
                "accuracy": 1.167,
                "latency": 285.705,
                "stderr": 1.065,
                "cost_per_test": 0.162773,
                "token_totals": {
                    "input_tokens": 12568177,
                    "output_tokens": 248245,
                    "reasoning_tokens": 240647,
                    "cache_read_tokens": 9619856,
                    "cache_write_tokens": 0
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Inception",
                "harness": null
            }
        }
    }
}