{
    "metadata": {
        "benchmark": "Finance Agent (v2)",
        "slug": "fabv2",
        "description": "Evaluating agents on core financial analyst tasks using the FAB v2 harness",
        "benchmark_id": "finance_agent_v2",
        "family": "finance_agent",
        "version": "2",
        "updated": "2026-10-07",
        "dataset_type": "private",
        "industry": "finance",
        "tasks": {
            "overall": "Overall",
            "all_pass": "All-Pass",
            "general_qualitative_analysis": "General Qualitative Analysis",
            "general_quantitative_analysis": "General Quantitative Analysis",
            "market_analysis": "Market Analysis",
            "comparables": "Comparables",
            "precedents": "Precedents",
            "adjustments": "Adjustments",
            "earnings_analysis": "Earnings Analysis",
            "disclosure_analysis": "Disclosure Analysis",
            "financial_modeling": "Financial Modeling"
        },
        "models": [
            "alibaba/qwen3.6-plus",
            "alibaba/qwen3.7-max",
            "alibaba/qwen3.7-plus",
            "alibaba/qwen3.8-27b",
            "alibaba/qwen3.8-max",
            "ant/ling-3.0-flash-2607",
            "ant/ling-3.0-flash-af-rc3",
            "anthropic/claude-fable-5",
            "anthropic/claude-fable-5-1",
            "anthropic/claude-haiku-4-5-20251001-thinking",
            "anthropic/claude-haiku-5-5",
            "anthropic/claude-opus-4-7",
            "anthropic/claude-opus-4-8",
            "anthropic/claude-opus-5",
            "anthropic/claude-opus-5-5",
            "anthropic/claude-sonnet-4-6",
            "anthropic/claude-sonnet-5",
            "anthropic/claude-sonnet-5-5",
            "cohere/command-a-plus-05-2026",
            "deepseek/deepseek-v4-flash-0731",
            "deepseek/deepseek-v4-pro",
            "deepseek/deepseek-v4-pro-0813",
            "deepseek/deepseek-v4.1-flash",
            "fireworks/ember-1",
            "fireworks/nemotron-lightning-3p5-30b-a3b",
            "google/gemini-3-flash-preview",
            "google/gemini-3.1-flash-lite-preview",
            "google/gemini-3.1-pro-preview",
            "google/gemini-3.5-flash",
            "google/gemini-3.5-flash-lite",
            "google/gemini-3.6-flash",
            "google/gemini-3.7-flash",
            "google/gemini-3.8-flash",
            "google/gemini-4-argon",
            "grok/grok-4.20-0309-reasoning",
            "grok/grok-4.3",
            "grok/grok-4.5",
            "grok/grok-4.6",
            "grok/grok-4.7",
            "inception/mercury-2.5",
            "kimi/kimi-k2.5-thinking",
            "kimi/kimi-k2.6",
            "kimi/kimi-k3",
            "meta/muse_spark_1_1",
            "meta/muse_spark_1_2",
            "meta/muse_spark_1_3",
            "meta/muse_spark_1_3_max",
            "minimax/MiniMax-M2.7",
            "minimax/MiniMax-M3",
            "mistralai/mistral-large-4",
            "mistralai/mistral-medium-3.5",
            "nvidia/nemotron-3-ultra-550b-a55b",
            "openai/gpt-5.4-mini-2026-03-17",
            "openai/gpt-5.4-nano-2026-03-17",
            "openai/gpt-5.5",
            "openai/gpt-5.6-luna",
            "openai/gpt-5.6-sol",
            "openai/gpt-5.6-terra",
            "openai/gpt-6-astra",
            "openai/gpt-6-luna",
            "openai/gpt-6-sol",
            "openai/gpt-6.1-sol",
            "poolside/laguna-m.1",
            "poolside/laguna-xs.2",
            "stepfun/step-5-preview",
            "tencent/hy4-preview",
            "thinkingmachines/inkling",
            "thinkingmachines/inkling-small",
            "xiaomi/mimo-v2.5",
            "xiaomi/mimo-v2.5-pro",
            "xiaomi/mimo-v2.6-flash",
            "xiaomi/mimo-v2.6-pro",
            "zai/glm-5.1",
            "zai/glm-5.2",
            "zai/glm-5.3",
            "zai/glm-5.3-flash"
        ],
        "partners": [],
        "showBadge": false,
        "visible": true,
        "use_cost_per_test": true,
        "runner": "external",
        "mode": "agentic",
        "archived": false,
        "partner": false,
        "total_models": 76
    },
    "tasks": {
        "overall": {
            "google/gemini-4-argon": {
                "accuracy": 65.401,
                "latency": 980.524,
                "stderr": 0.321,
                "cost_per_test": 4.378161,
                "token_totals": {
                    "input_tokens": 609679241,
                    "output_tokens": 31503850,
                    "reasoning_tokens": 13402361,
                    "cache_read_tokens": 289110948,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.8-flash": {
                "accuracy": 61.435,
                "latency": 200.952,
                "stderr": 0.128,
                "cost_per_test": 2.000457,
                "token_totals": {
                    "input_tokens": 1118010382,
                    "output_tokens": 25909358,
                    "reasoning_tokens": 18104089,
                    "cache_read_tokens": 719355707,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "meta/muse_spark_1_2": {
                "accuracy": 60.599,
                "latency": 304.205,
                "stderr": 0.279,
                "cost_per_test": 0.76972,
                "token_totals": {
                    "input_tokens": 424942458,
                    "output_tokens": 12218400,
                    "reasoning_tokens": 6559929,
                    "cache_read_tokens": null,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "meta/muse_spark_1_3_max": {
                "accuracy": 59.958,
                "latency": 198.929,
                "stderr": 2.06,
                "cost_per_test": 0.756512,
                "token_totals": {
                    "input_tokens": 508981355,
                    "output_tokens": 16233804,
                    "reasoning_tokens": 9783963,
                    "cache_read_tokens": 331627422,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "google/gemini-3.7-flash": {
                "accuracy": 59.042,
                "latency": 181.334,
                "stderr": 0.272,
                "cost_per_test": 1.477884,
                "token_totals": {
                    "input_tokens": 711083728,
                    "output_tokens": 16887053,
                    "reasoning_tokens": 10058516,
                    "cache_read_tokens": 391282104,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "meta/muse_spark_1_3": {
                "accuracy": 58.901,
                "latency": 341.936,
                "stderr": 0.444,
                "cost_per_test": 0.741132,
                "token_totals": {
                    "input_tokens": 413247670,
                    "output_tokens": 14548497,
                    "reasoning_tokens": 8843980,
                    "cache_read_tokens": null,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "anthropic/claude-fable-5-1": {
                "accuracy": 58.877,
                "latency": 1094.165,
                "stderr": 2.058,
                "cost_per_test": 8.346696,
                "token_totals": {
                    "input_tokens": 458174812,
                    "output_tokens": 33030168,
                    "reasoning_tokens": 19377315,
                    "cache_read_tokens": 228829928,
                    "cache_write_tokens": 229316981
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-5": {
                "accuracy": 58.633,
                "latency": 598.424,
                "stderr": 0.077,
                "cost_per_test": 5.124473,
                "token_totals": {
                    "input_tokens": 543793980,
                    "output_tokens": 22921224,
                    "reasoning_tokens": 9150666,
                    "cache_read_tokens": 253099566,
                    "cache_write_tokens": 290670709
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-5-5": {
                "accuracy": 58.587,
                "latency": 1987.81,
                "stderr": 0.17,
                "cost_per_test": 9.215734,
                "token_totals": {
                    "input_tokens": 763751733,
                    "output_tokens": 109304455,
                    "reasoning_tokens": 94811735,
                    "cache_read_tokens": 413590027,
                    "cache_write_tokens": 350117722
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-sonnet-5-5": {
                "accuracy": 58.103,
                "latency": 1903.336,
                "stderr": 0.673,
                "cost_per_test": 6.552431,
                "token_totals": {
                    "input_tokens": 1121526513,
                    "output_tokens": 137870326,
                    "reasoning_tokens": 120529788,
                    "cache_read_tokens": 555448124,
                    "cache_write_tokens": 566027579
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.5-flash": {
                "accuracy": 57.861,
                "latency": 322.279,
                "stderr": 0.231,
                "cost_per_test": 2.505488,
                "token_totals": {
                    "input_tokens": 1227406624,
                    "output_tokens": 26782944,
                    "reasoning_tokens": 20834822,
                    "cache_read_tokens": 707613227,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "zai/glm-5.3-flash": {
                "accuracy": 57.85,
                "latency": 890.519,
                "stderr": 2.061,
                "cost_per_test": 0.047625,
                "token_totals": {
                    "input_tokens": 333663746,
                    "output_tokens": 26908803,
                    "reasoning_tokens": null,
                    "cache_read_tokens": 172014080,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "xiaomi/mimo-v2.6-pro": {
                "accuracy": 57.339,
                "latency": 648.075,
                "stderr": 0.574,
                "cost_per_test": 0.198909,
                "token_totals": {
                    "input_tokens": 305876104,
                    "output_tokens": 13378019,
                    "reasoning_tokens": 7659784,
                    "cache_read_tokens": 127922645,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "meta/muse_spark_1_1": {
                "accuracy": 57.207,
                "latency": 414.77,
                "stderr": 0.789,
                "cost_per_test": 0.723391,
                "token_totals": {
                    "input_tokens": 363982665,
                    "output_tokens": 15279252,
                    "reasoning_tokens": 10574096,
                    "cache_read_tokens": 177594427,
                    "cache_write_tokens": null
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "anthropic/claude-fable-5": {
                "accuracy": 56.314,
                "latency": 610.952,
                "stderr": 0.84,
                "cost_per_test": 8.061563,
                "token_totals": {
                    "input_tokens": 311882539,
                    "output_tokens": 19969523,
                    "reasoning_tokens": null,
                    "cache_read_tokens": 110357130,
                    "cache_write_tokens": 201446408
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.6-flash": {
                "accuracy": 56.296,
                "latency": 224.899,
                "stderr": 0.183,
                "cost_per_test": 1.402901,
                "token_totals": {
                    "input_tokens": 627928399,
                    "output_tokens": 16297009,
                    "reasoning_tokens": 11292206,
                    "cache_read_tokens": 320603458,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "xiaomi/mimo-v2.6-flash": {
                "accuracy": 56.277,
                "latency": 421.99,
                "stderr": 0.462,
                "cost_per_test": 0.071536,
                "token_totals": {
                    "input_tokens": 349885254,
                    "output_tokens": 15078803,
                    "reasoning_tokens": 7475980,
                    "cache_read_tokens": 154212352,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "zai/glm-5.3": {
                "accuracy": 55.84,
                "latency": 949.539,
                "stderr": 2.066,
                "cost_per_test": 1.072768,
                "token_totals": {
                    "input_tokens": 363886559,
                    "output_tokens": 36729670,
                    "reasoning_tokens": null,
                    "cache_read_tokens": null,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "tencent/hy4-preview": {
                "accuracy": 55.063,
                "latency": 1424.632,
                "stderr": 0.308,
                "cost_per_test": 0.5903,
                "token_totals": {
                    "input_tokens": 316560636,
                    "output_tokens": 41494337,
                    "reasoning_tokens": 34844249,
                    "cache_read_tokens": 129727253,
                    "cache_write_tokens": 0
                },
                "temperature": 0.9,
                "top_p": 1.0,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Tencent",
                "harness": null
            },
            "openai/gpt-5.6-luna": {
                "accuracy": 55.044,
                "latency": 772.153,
                "stderr": 0.309,
                "cost_per_test": 0.279224,
                "token_totals": {
                    "input_tokens": 533549683,
                    "output_tokens": 52574268,
                    "reasoning_tokens": 45760878,
                    "cache_read_tokens": null,
                    "cache_write_tokens": null
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "ant/ling-3.0-flash-af-rc3": {
                "accuracy": 54.927,
                "latency": 375.066,
                "stderr": 0.679,
                "cost_per_test": 0.044792,
                "token_totals": {
                    "input_tokens": 401804903,
                    "output_tokens": 19410118,
                    "reasoning_tokens": null,
                    "cache_read_tokens": 155120896,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "mistralai/mistral-large-4": {
                "accuracy": 54.678,
                "latency": 1561.879,
                "stderr": 0.58,
                "cost_per_test": 1.201399,
                "token_totals": {
                    "input_tokens": 365895891,
                    "output_tokens": 27874175,
                    "reasoning_tokens": null,
                    "cache_read_tokens": 60248405,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "openai/gpt-5.6-terra": {
                "accuracy": 54.436,
                "latency": 1449.852,
                "stderr": 2.071,
                "cost_per_test": 3.629197,
                "token_totals": {
                    "input_tokens": 651703808,
                    "output_tokens": 67770045,
                    "reasoning_tokens": 58991995,
                    "cache_read_tokens": 334920434,
                    "cache_write_tokens": null
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-haiku-5-5": {
                "accuracy": 54.138,
                "latency": 1849.0,
                "stderr": 2.132,
                "cost_per_test": 1.636603,
                "token_totals": {
                    "input_tokens": 1125304863,
                    "output_tokens": 207972605,
                    "reasoning_tokens": 195320438,
                    "cache_read_tokens": 434937131,
                    "cache_write_tokens": 690320292
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-4-8": {
                "accuracy": 53.918,
                "latency": 547.397,
                "stderr": 0.159,
                "cost_per_test": 4.218538,
                "token_totals": {
                    "input_tokens": 309993285,
                    "output_tokens": 17301258,
                    "reasoning_tokens": null,
                    "cache_read_tokens": 82013466,
                    "cache_write_tokens": 227923817
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-sonnet-5": {
                "accuracy": 53.909,
                "latency": 792.086,
                "stderr": 0.518,
                "cost_per_test": 0.750759,
                "token_totals": {
                    "input_tokens": 164610121,
                    "output_tokens": 10102005,
                    "reasoning_tokens": null,
                    "cache_read_tokens": 75956629,
                    "cache_write_tokens": 88646581
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.6-sol": {
                "accuracy": 53.756,
                "latency": 1165.258,
                "stderr": 0.846,
                "cost_per_test": 1.253419,
                "token_totals": {
                    "input_tokens": 322058918,
                    "output_tokens": 6216567,
                    "reasoning_tokens": 4178719,
                    "cache_read_tokens": 276042752,
                    "cache_write_tokens": null
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4.6": {
                "accuracy": 53.682,
                "latency": 1000.937,
                "stderr": 0.67,
                "cost_per_test": 1.657732,
                "token_totals": {
                    "input_tokens": 495041785,
                    "output_tokens": 24551729,
                    "reasoning_tokens": 16279727,
                    "cache_read_tokens": 260943147,
                    "cache_write_tokens": null
                },
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-6-astra": {
                "accuracy": 53.54,
                "latency": 692.672,
                "stderr": 2.078,
                "cost_per_test": 6.81746,
                "token_totals": {
                    "input_tokens": 240568066,
                    "output_tokens": 22397666,
                    "reasoning_tokens": 16813955,
                    "cache_read_tokens": 70591920,
                    "cache_write_tokens": null
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "deepseek/deepseek-v4.1-flash": {
                "accuracy": 53.481,
                "latency": 364.376,
                "stderr": 0.39,
                "cost_per_test": 0.207465,
                "token_totals": {
                    "input_tokens": 354713236,
                    "output_tokens": 33772547,
                    "reasoning_tokens": 28648435,
                    "cache_read_tokens": 182250763,
                    "cache_write_tokens": null
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "kimi/kimi-k3": {
                "accuracy": 53.114,
                "latency": 285.753,
                "stderr": 0.374,
                "cost_per_test": 1.913224,
                "token_totals": {
                    "input_tokens": 460080913,
                    "output_tokens": 16314784,
                    "reasoning_tokens": 10722175,
                    "cache_read_tokens": 282968006,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "grok/grok-4.7": {
                "accuracy": 52.251,
                "latency": 932.724,
                "stderr": 0.355,
                "cost_per_test": 2.719136,
                "token_totals": {
                    "input_tokens": 827415903,
                    "output_tokens": 23428536,
                    "reasoning_tokens": 15352966,
                    "cache_read_tokens": 444856320,
                    "cache_write_tokens": null
                },
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-6.1-sol": {
                "accuracy": 52.033,
                "latency": 925.46,
                "stderr": 0.308,
                "cost_per_test": 1.619399,
                "token_totals": {
                    "input_tokens": 256899959,
                    "output_tokens": 22298636,
                    "reasoning_tokens": 16662434,
                    "cache_read_tokens": 75246545,
                    "cache_write_tokens": 181603278
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "fireworks/ember-1": {
                "accuracy": 51.853,
                "latency": 798.612,
                "stderr": 2.124,
                "cost_per_test": 1.95445,
                "token_totals": {
                    "input_tokens": 435525965,
                    "output_tokens": 13497155,
                    "reasoning_tokens": 8350801,
                    "cache_read_tokens": 233160321,
                    "cache_write_tokens": null
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 51.76,
                "latency": 662.269,
                "stderr": 0.55,
                "cost_per_test": 4.149702,
                "token_totals": {
                    "input_tokens": 423716492,
                    "output_tokens": 17794792,
                    "reasoning_tokens": 13531292,
                    "cache_read_tokens": 217525163,
                    "cache_write_tokens": null
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-4-7": {
                "accuracy": 51.509,
                "latency": 360.437,
                "stderr": 0.49,
                "cost_per_test": 4.032269,
                "token_totals": {
                    "input_tokens": 370221415,
                    "output_tokens": 12293417,
                    "reasoning_tokens": null,
                    "cache_read_tokens": 140274986,
                    "cache_write_tokens": 229852824
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 51.035,
                "latency": 695.588,
                "stderr": 0.329,
                "cost_per_test": 2.409559,
                "token_totals": {
                    "input_tokens": 421005921,
                    "output_tokens": 20173634,
                    "reasoning_tokens": null,
                    "cache_read_tokens": 230989311,
                    "cache_write_tokens": 189800855
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "stepfun/step-5-preview": {
                "accuracy": 50.671,
                "latency": 747.585,
                "stderr": 0.324,
                "cost_per_test": 0.619283,
                "token_totals": {
                    "input_tokens": 464025197,
                    "output_tokens": 20770861,
                    "reasoning_tokens": 0,
                    "cache_read_tokens": 254565035,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 1024000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Stepfun",
                "harness": null
            },
            "alibaba/qwen3.8-max": {
                "accuracy": 50.593,
                "latency": 1174.928,
                "stderr": 0.578,
                "cost_per_test": 1.236977,
                "token_totals": {
                    "input_tokens": 313950822,
                    "output_tokens": 28653788,
                    "reasoning_tokens": 23202069,
                    "cache_read_tokens": 139669632,
                    "cache_write_tokens": null
                },
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "deepseek/deepseek-v4-pro-0813": {
                "accuracy": 50.393,
                "latency": 974.518,
                "stderr": 0.249,
                "cost_per_test": 0.880465,
                "token_totals": {
                    "input_tokens": 419761736,
                    "output_tokens": 30621918,
                    "reasoning_tokens": 25089786,
                    "cache_read_tokens": 218761173,
                    "cache_write_tokens": null
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "openai/gpt-6-luna": {
                "accuracy": 49.873,
                "latency": 897.881,
                "stderr": 0.232,
                "cost_per_test": 0.124338,
                "token_totals": {
                    "input_tokens": 413604271,
                    "output_tokens": 45731435,
                    "reasoning_tokens": 41794427,
                    "cache_read_tokens": 199389488,
                    "cache_write_tokens": 214114651
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-5.2": {
                "accuracy": 49.699,
                "latency": 473.158,
                "stderr": 0.878,
                "cost_per_test": 0.713981,
                "token_totals": {
                    "input_tokens": 271639159,
                    "output_tokens": 12183429,
                    "reasoning_tokens": 7820405,
                    "cache_read_tokens": 98781099,
                    "cache_write_tokens": null
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "deepseek/deepseek-v4-flash-0731": {
                "accuracy": 49.517,
                "latency": 490.245,
                "stderr": 0.465,
                "cost_per_test": 0.276154,
                "token_totals": {
                    "input_tokens": 406112640,
                    "output_tokens": 27423276,
                    "reasoning_tokens": 22726163,
                    "cache_read_tokens": 213584768,
                    "cache_write_tokens": null
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "openai/gpt-6-sol": {
                "accuracy": 49.05,
                "latency": 539.907,
                "stderr": 0.58,
                "cost_per_test": 2.122121,
                "token_totals": {
                    "input_tokens": 416329027,
                    "output_tokens": 22746494,
                    "reasoning_tokens": 18774690,
                    "cache_read_tokens": 155710451,
                    "cache_write_tokens": 260528040
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3.8-27b": {
                "accuracy": 48.553,
                "latency": 621.733,
                "stderr": 2.096,
                "cost_per_test": 0.748532,
                "token_totals": {
                    "input_tokens": 422396163,
                    "output_tokens": 41880461,
                    "reasoning_tokens": 34693358,
                    "cache_read_tokens": 0,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "grok/grok-4.5": {
                "accuracy": 48.349,
                "latency": 289.014,
                "stderr": 0.458,
                "cost_per_test": 1.156495,
                "token_totals": {
                    "input_tokens": 346429154,
                    "output_tokens": 13008523,
                    "reasoning_tokens": 7208719,
                    "cache_read_tokens": 193601579,
                    "cache_write_tokens": null
                },
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "minimax/MiniMax-M3": {
                "accuracy": 48.269,
                "latency": 496.894,
                "stderr": 0.437,
                "cost_per_test": 0.32345,
                "token_totals": {
                    "input_tokens": 302569216,
                    "output_tokens": 10069655,
                    "reasoning_tokens": null,
                    "cache_read_tokens": 126041678,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 512000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "alibaba/qwen3.7-max": {
                "accuracy": 47.776,
                "latency": 635.178,
                "stderr": 0.727,
                "cost_per_test": 0.553214,
                "token_totals": {
                    "input_tokens": 104973358,
                    "output_tokens": 3989677,
                    "reasoning_tokens": 2814264,
                    "cache_read_tokens": 19536384,
                    "cache_write_tokens": null
                },
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "google/gemini-3.5-flash-lite": {
                "accuracy": 47.442,
                "latency": 142.913,
                "stderr": 0.513,
                "cost_per_test": 0.389575,
                "token_totals": {
                    "input_tokens": 670526775,
                    "output_tokens": 10900400,
                    "reasoning_tokens": 7520808,
                    "cache_read_tokens": 196667284,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "thinkingmachines/inkling": {
                "accuracy": 46.597,
                "latency": 1082.125,
                "stderr": 0.925,
                "cost_per_test": 0.996264,
                "token_totals": {
                    "input_tokens": 882963514,
                    "output_tokens": 24911364,
                    "reasoning_tokens": null,
                    "cache_read_tokens": 647224491,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 45.36,
                "latency": 2153.503,
                "stderr": 0.447,
                "cost_per_test": 1.201815,
                "token_totals": {
                    "input_tokens": 609985753,
                    "output_tokens": 72941360,
                    "reasoning_tokens": 70078573,
                    "cache_read_tokens": 362827776,
                    "cache_write_tokens": null
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 44.899,
                "latency": 1629.639,
                "stderr": 0.733,
                "cost_per_test": 1.141587,
                "token_totals": {
                    "input_tokens": 846905588,
                    "output_tokens": 23540852,
                    "reasoning_tokens": 0,
                    "cache_read_tokens": 488314911,
                    "cache_write_tokens": null
                },
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "zai/glm-5.1": {
                "accuracy": 44.792,
                "latency": 924.078,
                "stderr": 0.606,
                "cost_per_test": 0.618185,
                "token_totals": {
                    "input_tokens": 498811983,
                    "output_tokens": 15910910,
                    "reasoning_tokens": 12077030,
                    "cache_read_tokens": 339429333,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 44.083,
                "latency": 1526.183,
                "stderr": 0.653,
                "cost_per_test": 0.879086,
                "token_totals": {
                    "input_tokens": 508078472,
                    "output_tokens": 22058372,
                    "reasoning_tokens": 17233163,
                    "cache_read_tokens": 284032811,
                    "cache_write_tokens": null
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 42.982,
                "latency": 208.258,
                "stderr": 1.206,
                "cost_per_test": 1.591906,
                "token_totals": {
                    "input_tokens": 365193664,
                    "output_tokens": 7730374,
                    "reasoning_tokens": 5787767,
                    "cache_read_tokens": 123797501,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 42.551,
                "latency": 246.543,
                "stderr": 0.434,
                "cost_per_test": 0.570657,
                "token_totals": {
                    "input_tokens": 657633680,
                    "output_tokens": 15218094,
                    "reasoning_tokens": 12099810,
                    "cache_read_tokens": 261501224,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "xiaomi/mimo-v2.5-pro": {
                "accuracy": 41.502,
                "latency": 497.373,
                "stderr": 1.193,
                "cost_per_test": 0.206445,
                "token_totals": {
                    "input_tokens": 507567154,
                    "output_tokens": 8732144,
                    "reasoning_tokens": 5054819,
                    "cache_read_tokens": 314066453,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "thinkingmachines/inkling-small": {
                "accuracy": 41.257,
                "latency": 994.255,
                "stderr": 0.832,
                "cost_per_test": 0.230271,
                "token_totals": {
                    "input_tokens": 696484546,
                    "output_tokens": 28023249,
                    "reasoning_tokens": null,
                    "cache_read_tokens": 580562347,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 262000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 40.846,
                "latency": 611.907,
                "stderr": 0.131,
                "cost_per_test": 0.814379,
                "token_totals": {
                    "input_tokens": 488661479,
                    "output_tokens": 11569301,
                    "reasoning_tokens": 8412225,
                    "cache_read_tokens": 0,
                    "cache_write_tokens": null
                },
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "alibaba/qwen3.7-plus": {
                "accuracy": 38.22,
                "latency": 542.045,
                "stderr": 1.04,
                "cost_per_test": 0.357661,
                "token_totals": {
                    "input_tokens": 446705484,
                    "output_tokens": 9996981,
                    "reasoning_tokens": 7272392,
                    "cache_read_tokens": 192899072,
                    "cache_write_tokens": null
                },
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 38.217,
                "latency": 334.538,
                "stderr": 1.185,
                "cost_per_test": 0.15925,
                "token_totals": {
                    "input_tokens": 584481947,
                    "output_tokens": 10638763,
                    "reasoning_tokens": 7882802,
                    "cache_read_tokens": 325180587,
                    "cache_write_tokens": null
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4.3": {
                "accuracy": 37.728,
                "latency": 615.499,
                "stderr": 0.373,
                "cost_per_test": 0.848345,
                "token_totals": {
                    "input_tokens": 348106276,
                    "output_tokens": 15533931,
                    "reasoning_tokens": 12549303,
                    "cache_read_tokens": 125272960,
                    "cache_write_tokens": null
                },
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "nvidia/nemotron-3-ultra-550b-a55b": {
                "accuracy": 37.667,
                "latency": 253.882,
                "stderr": 0.149,
                "cost_per_test": null,
                "token_totals": {
                    "input_tokens": 408960062,
                    "output_tokens": 7705379,
                    "reasoning_tokens": 0,
                    "cache_read_tokens": 0,
                    "cache_write_tokens": null
                },
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Nvidia",
                "harness": null
            },
            "xiaomi/mimo-v2.5": {
                "accuracy": 36.728,
                "latency": 339.26,
                "stderr": 0.109,
                "cost_per_test": 0.08693,
                "token_totals": {
                    "input_tokens": 616264480,
                    "output_tokens": 8440426,
                    "reasoning_tokens": 5101469,
                    "cache_read_tokens": 360948032,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 35.792,
                "latency": 751.754,
                "stderr": 1.989,
                "cost_per_test": 0.294389,
                "token_totals": {
                    "input_tokens": 324471186,
                    "output_tokens": 7830906,
                    "reasoning_tokens": 5301673,
                    "cache_read_tokens": 171400704,
                    "cache_write_tokens": null
                },
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "mistralai/mistral-medium-3.5": {
                "accuracy": 32.103,
                "latency": 768.405,
                "stderr": 0.967,
                "cost_per_test": 1.908513,
                "token_totals": {
                    "input_tokens": 448123641,
                    "output_tokens": 24450234,
                    "reasoning_tokens": null,
                    "cache_read_tokens": null,
                    "cache_write_tokens": null
                },
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 31.01,
                "latency": 176.769,
                "stderr": 0.564,
                "cost_per_test": 0.602129,
                "token_totals": {
                    "input_tokens": 314188143,
                    "output_tokens": 7984201,
                    "reasoning_tokens": null,
                    "cache_read_tokens": 140237697,
                    "cache_write_tokens": 172251987
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "ant/ling-3.0-flash-2607": {
                "accuracy": 30.314,
                "latency": 157.02,
                "stderr": 0.275,
                "cost_per_test": 0.05047,
                "token_totals": {
                    "input_tokens": 593259509,
                    "output_tokens": 13185323,
                    "reasoning_tokens": null,
                    "cache_read_tokens": 426536000,
                    "cache_write_tokens": null
                },
                "temperature": 0.6,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 29.988,
                "latency": 54.778,
                "stderr": 0.643,
                "cost_per_test": 0.142482,
                "token_totals": {
                    "input_tokens": 309249776,
                    "output_tokens": 4518540,
                    "reasoning_tokens": 3033245,
                    "cache_read_tokens": 88769598,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 28.492,
                "latency": 351.413,
                "stderr": 0.315,
                "cost_per_test": 0.791809,
                "token_totals": {
                    "input_tokens": 181695255,
                    "output_tokens": 12341665,
                    "reasoning_tokens": 9833132,
                    "cache_read_tokens": 66807493,
                    "cache_write_tokens": null
                },
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 27.887,
                "latency": 803.528,
                "stderr": 0.508,
                "cost_per_test": 0.18348,
                "token_totals": {
                    "input_tokens": 573686662,
                    "output_tokens": 15857919,
                    "reasoning_tokens": null,
                    "cache_read_tokens": 404654261,
                    "cache_write_tokens": 9164196
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "poolside/laguna-m.1": {
                "accuracy": 25.026,
                "latency": 1223.035,
                "stderr": 1.277,
                "cost_per_test": 0.818192,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "inception/mercury-2.5": {
                "accuracy": 18.894,
                "latency": 92.482,
                "stderr": 0.216,
                "cost_per_test": 0.106441,
                "token_totals": {
                    "input_tokens": 376083870,
                    "output_tokens": 5831330,
                    "reasoning_tokens": 4955938,
                    "cache_read_tokens": 191920653,
                    "cache_write_tokens": null
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Inception",
                "harness": null
            },
            "fireworks/nemotron-lightning-3p5-30b-a3b": {
                "accuracy": 18.5,
                "latency": 539.61,
                "stderr": 0.882,
                "cost_per_test": 0.028554,
                "token_totals": {
                    "input_tokens": 335136335,
                    "output_tokens": 19277054,
                    "reasoning_tokens": null,
                    "cache_read_tokens": null,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "poolside/laguna-xs.2": {
                "accuracy": 15.601,
                "latency": 526.624,
                "stderr": 0.326,
                "cost_per_test": 0.19216,
                "token_totals": {
                    "input_tokens": 1269502184,
                    "output_tokens": 8838931,
                    "reasoning_tokens": 5301032,
                    "cache_read_tokens": 283700016,
                    "cache_write_tokens": 0
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "cohere/command-a-plus-05-2026": {
                "accuracy": 9.044,
                "latency": 566.745,
                "stderr": 0.409,
                "cost_per_test": 1.63003,
                "token_totals": {
                    "input_tokens": 212326507,
                    "output_tokens": 20269701,
                    "reasoning_tokens": 19053167,
                    "cache_read_tokens": 0,
                    "cache_write_tokens": null
                },
                "temperature": 0.9,
                "top_p": 0.95,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            }
        },
        "adjustments": {
            "google/gemini-4-argon": {
                "accuracy": 60.527,
                "latency": 911.574,
                "stderr": 2.143,
                "cost_per_test": 4.200264,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "meta/muse_spark_1_2": {
                "accuracy": 56.288,
                "latency": 319.907,
                "stderr": 0.465,
                "cost_per_test": 0.705031,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "google/gemini-3.5-flash": {
                "accuracy": 54.869,
                "latency": 317.205,
                "stderr": 1.14,
                "cost_per_test": 2.304186,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.8-flash": {
                "accuracy": 54.828,
                "latency": 203.646,
                "stderr": 2.403,
                "cost_per_test": 1.828394,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.7-flash": {
                "accuracy": 54.718,
                "latency": 171.627,
                "stderr": 0.739,
                "cost_per_test": 1.506148,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-5": {
                "accuracy": 54.617,
                "latency": 538.252,
                "stderr": 0.815,
                "cost_per_test": 5.209947,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-5-5": {
                "accuracy": 54.528,
                "latency": 1878.966,
                "stderr": 0.44,
                "cost_per_test": 8.970204,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-fable-5-1": {
                "accuracy": 54.393,
                "latency": 765.835,
                "stderr": 6.935,
                "cost_per_test": 7.936173,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.6-flash": {
                "accuracy": 53.608,
                "latency": 411.19,
                "stderr": 1.416,
                "cost_per_test": 1.321839,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "zai/glm-5.3-flash": {
                "accuracy": 53.193,
                "latency": 823.579,
                "stderr": 7.034,
                "cost_per_test": 0.043023,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "meta/muse_spark_1_3": {
                "accuracy": 52.818,
                "latency": 297.402,
                "stderr": 0.09,
                "cost_per_test": 0.667124,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "anthropic/claude-sonnet-5-5": {
                "accuracy": 52.746,
                "latency": 1831.95,
                "stderr": 2.119,
                "cost_per_test": 6.302773,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "meta/muse_spark_1_3_max": {
                "accuracy": 52.726,
                "latency": 183.755,
                "stderr": 6.993,
                "cost_per_test": 0.787963,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "zai/glm-5.3": {
                "accuracy": 52.726,
                "latency": 1166.284,
                "stderr": 6.993,
                "cost_per_test": 1.069415,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "mistralai/mistral-large-4": {
                "accuracy": 52.429,
                "latency": 1403.285,
                "stderr": 1.277,
                "cost_per_test": 1.138787,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "openai/gpt-5.6-luna": {
                "accuracy": 52.027,
                "latency": 769.98,
                "stderr": 1.148,
                "cost_per_test": 0.269986,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "ant/ling-3.0-flash-af-rc3": {
                "accuracy": 51.807,
                "latency": 375.07,
                "stderr": 1.582,
                "cost_per_test": 0.039694,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "xiaomi/mimo-v2.6-pro": {
                "accuracy": 51.572,
                "latency": 581.6,
                "stderr": 2.4,
                "cost_per_test": 0.188041,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "xiaomi/mimo-v2.6-flash": {
                "accuracy": 51.35,
                "latency": 404.881,
                "stderr": 0.046,
                "cost_per_test": 0.071338,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "fireworks/ember-1": {
                "accuracy": 51.258,
                "latency": 679.076,
                "stderr": 7.044,
                "cost_per_test": 1.714992,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "anthropic/claude-fable-5": {
                "accuracy": 51.092,
                "latency": 644.817,
                "stderr": 0.338,
                "cost_per_test": 7.417657,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-6-luna": {
                "accuracy": 49.75,
                "latency": 884.535,
                "stderr": 0.542,
                "cost_per_test": 0.114148,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.6-terra": {
                "accuracy": 49.711,
                "latency": 1517.352,
                "stderr": 6.902,
                "cost_per_test": 3.890731,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "meta/muse_spark_1_1": {
                "accuracy": 49.66,
                "latency": 331.405,
                "stderr": 2.337,
                "cost_per_test": 0.707474,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "grok/grok-4.6": {
                "accuracy": 49.619,
                "latency": 755.354,
                "stderr": 0.897,
                "cost_per_test": 1.422811,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-haiku-5-5": {
                "accuracy": 49.557,
                "latency": 1778.73,
                "stderr": 7.086,
                "cost_per_test": 1.513789,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "deepseek/deepseek-v4.1-flash": {
                "accuracy": 49.241,
                "latency": 356.743,
                "stderr": 0.989,
                "cost_per_test": 0.202031,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "openai/gpt-5.6-sol": {
                "accuracy": 49.233,
                "latency": 1135.714,
                "stderr": 0.804,
                "cost_per_test": 0.941562,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 48.92,
                "latency": 605.669,
                "stderr": 1.314,
                "cost_per_test": 2.175111,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "deepseek/deepseek-v4-pro-0813": {
                "accuracy": 48.801,
                "latency": 1056.943,
                "stderr": 1.156,
                "cost_per_test": 0.943878,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "tencent/hy4-preview": {
                "accuracy": 48.771,
                "latency": 1355.541,
                "stderr": 2.442,
                "cost_per_test": 0.541966,
                "temperature": 0.9,
                "top_p": 1.0,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Tencent",
                "harness": null
            },
            "anthropic/claude-sonnet-5": {
                "accuracy": 48.715,
                "latency": 786.124,
                "stderr": 0.375,
                "cost_per_test": 2.353391,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3.8-max": {
                "accuracy": 48.426,
                "latency": 1219.23,
                "stderr": 1.532,
                "cost_per_test": 1.129108,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "anthropic/claude-opus-4-8": {
                "accuracy": 48.387,
                "latency": 705.323,
                "stderr": 2.53,
                "cost_per_test": 3.790165,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-6-astra": {
                "accuracy": 48.318,
                "latency": 592.224,
                "stderr": 6.95,
                "cost_per_test": 6.103254,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "deepseek/deepseek-v4-flash-0731": {
                "accuracy": 47.819,
                "latency": 504.554,
                "stderr": 0.983,
                "cost_per_test": 0.290301,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "stepfun/step-5-preview": {
                "accuracy": 47.752,
                "latency": 1058.392,
                "stderr": 2.656,
                "cost_per_test": 0.603494,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 1024000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Stepfun",
                "harness": null
            },
            "alibaba/qwen3.8-27b": {
                "accuracy": 47.345,
                "latency": 691.511,
                "stderr": 7.051,
                "cost_per_test": 0.827023,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "zai/glm-5.2": {
                "accuracy": 46.305,
                "latency": 410.892,
                "stderr": 2.981,
                "cost_per_test": 0.619233,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "grok/grok-4.7": {
                "accuracy": 46.25,
                "latency": 847.684,
                "stderr": 1.621,
                "cost_per_test": 2.145058,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 45.74,
                "latency": 651.473,
                "stderr": 1.121,
                "cost_per_test": 3.890026,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-4-7": {
                "accuracy": 45.366,
                "latency": 294.357,
                "stderr": 0.975,
                "cost_per_test": 3.219143,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 45.366,
                "latency": 2520.252,
                "stderr": 1.159,
                "cost_per_test": 1.132987,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "kimi/kimi-k3": {
                "accuracy": 45.274,
                "latency": 258.53,
                "stderr": 1.417,
                "cost_per_test": 1.887778,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "openai/gpt-6.1-sol": {
                "accuracy": 44.717,
                "latency": 829.094,
                "stderr": 0.708,
                "cost_per_test": 1.492087,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-5.1": {
                "accuracy": 43.276,
                "latency": 936.369,
                "stderr": 3.152,
                "cost_per_test": 0.586714,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "openai/gpt-6-sol": {
                "accuracy": 43.241,
                "latency": 541.242,
                "stderr": 1.208,
                "cost_per_test": 2.023928,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "minimax/MiniMax-M3": {
                "accuracy": 42.321,
                "latency": 436.409,
                "stderr": 2.357,
                "cost_per_test": 0.280596,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 512000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "google/gemini-3.5-flash-lite": {
                "accuracy": 41.741,
                "latency": 477.075,
                "stderr": 0.645,
                "cost_per_test": 0.363606,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 41.586,
                "latency": 1486.77,
                "stderr": 1.185,
                "cost_per_test": 0.923173,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "alibaba/qwen3.7-max": {
                "accuracy": 41.176,
                "latency": 734.997,
                "stderr": 0.376,
                "cost_per_test": 1.959601,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "grok/grok-4.5": {
                "accuracy": 38.568,
                "latency": 234.84,
                "stderr": 1.902,
                "cost_per_test": 0.943055,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 38.361,
                "latency": 188.406,
                "stderr": 1.026,
                "cost_per_test": 1.417992,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 37.934,
                "latency": 1483.007,
                "stderr": 2.645,
                "cost_per_test": 1.064336,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "thinkingmachines/inkling": {
                "accuracy": 37.836,
                "latency": 1498.605,
                "stderr": 1.673,
                "cost_per_test": 0.942244,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 35.931,
                "latency": 226.691,
                "stderr": 1.014,
                "cost_per_test": 0.50541,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 35.673,
                "latency": 674.622,
                "stderr": 2.924,
                "cost_per_test": 0.666331,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "alibaba/qwen3.7-plus": {
                "accuracy": 35.538,
                "latency": 733.145,
                "stderr": 2.485,
                "cost_per_test": 0.424503,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "nvidia/nemotron-3-ultra-550b-a55b": {
                "accuracy": 34.449,
                "latency": 255.224,
                "stderr": 3.045,
                "cost_per_test": null,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Nvidia",
                "harness": null
            },
            "xiaomi/mimo-v2.5-pro": {
                "accuracy": 34.386,
                "latency": 489.296,
                "stderr": 1.658,
                "cost_per_test": 0.206068,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "thinkingmachines/inkling-small": {
                "accuracy": 34.0,
                "latency": 1123.995,
                "stderr": 4.474,
                "cost_per_test": 0.270302,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 262000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 33.418,
                "latency": 834.3,
                "stderr": 6.665,
                "cost_per_test": 0.296491,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 32.97,
                "latency": 360.491,
                "stderr": 0.667,
                "cost_per_test": 0.168389,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4.3": {
                "accuracy": 31.863,
                "latency": 677.894,
                "stderr": 2.297,
                "cost_per_test": 0.876016,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "xiaomi/mimo-v2.5": {
                "accuracy": 31.746,
                "latency": 328.95,
                "stderr": 2.451,
                "cost_per_test": 0.088997,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 30.969,
                "latency": 61.502,
                "stderr": 4.238,
                "cost_per_test": 0.177463,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 30.462,
                "latency": 182.053,
                "stderr": 0.789,
                "cost_per_test": 0.675348,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "mistralai/mistral-medium-3.5": {
                "accuracy": 27.312,
                "latency": 853.326,
                "stderr": 2.309,
                "cost_per_test": 1.995722,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "ant/ling-3.0-flash-2607": {
                "accuracy": 23.745,
                "latency": 206.49,
                "stderr": 2.38,
                "cost_per_test": 0.068504,
                "temperature": 0.6,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 20.972,
                "latency": 931.225,
                "stderr": 3.111,
                "cost_per_test": 0.213775,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 17.428,
                "latency": 504.099,
                "stderr": 2.564,
                "cost_per_test": 0.95371,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "poolside/laguna-m.1": {
                "accuracy": 15.759,
                "latency": 1541.799,
                "stderr": 0.61,
                "cost_per_test": 0.893142,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "fireworks/nemotron-lightning-3p5-30b-a3b": {
                "accuracy": 14.018,
                "latency": 731.655,
                "stderr": 0.949,
                "cost_per_test": 0.027811,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "inception/mercury-2.5": {
                "accuracy": 12.415,
                "latency": 177.236,
                "stderr": 0.396,
                "cost_per_test": 0.208655,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Inception",
                "harness": null
            },
            "poolside/laguna-xs.2": {
                "accuracy": 9.577,
                "latency": 663.412,
                "stderr": 0.889,
                "cost_per_test": 0.207781,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "cohere/command-a-plus-05-2026": {
                "accuracy": 3.107,
                "latency": 689.407,
                "stderr": 1.55,
                "cost_per_test": 1.978852,
                "temperature": 0.9,
                "top_p": 0.95,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            }
        },
        "disclosure_analysis": {
            "google/gemini-4-argon": {
                "accuracy": 74.496,
                "latency": 976.865,
                "stderr": 0.721,
                "cost_per_test": 4.421166,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "zai/glm-5.3-flash": {
                "accuracy": 71.543,
                "latency": 1564.901,
                "stderr": 6.112,
                "cost_per_test": 0.067904,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "anthropic/claude-sonnet-5-5": {
                "accuracy": 71.432,
                "latency": 2629.004,
                "stderr": 0.545,
                "cost_per_test": 9.050797,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-5": {
                "accuracy": 71.25,
                "latency": 1043.265,
                "stderr": 0.841,
                "cost_per_test": 7.184937,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-5-5": {
                "accuracy": 71.045,
                "latency": 2661.383,
                "stderr": 1.426,
                "cost_per_test": 12.462996,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "xiaomi/mimo-v2.6-flash": {
                "accuracy": 70.915,
                "latency": 467.049,
                "stderr": 1.067,
                "cost_per_test": 0.090017,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "anthropic/claude-fable-5-1": {
                "accuracy": 70.267,
                "latency": 1127.75,
                "stderr": 6.285,
                "cost_per_test": 10.886361,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "meta/muse_spark_1_3_max": {
                "accuracy": 70.079,
                "latency": 187.857,
                "stderr": 6.28,
                "cost_per_test": 0.834997,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "zai/glm-5.3": {
                "accuracy": 69.597,
                "latency": 1564.212,
                "stderr": 6.253,
                "cost_per_test": 1.404179,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "meta/muse_spark_1_2": {
                "accuracy": 69.489,
                "latency": 298.381,
                "stderr": 1.132,
                "cost_per_test": 0.904938,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "tencent/hy4-preview": {
                "accuracy": 68.844,
                "latency": 1782.48,
                "stderr": 1.216,
                "cost_per_test": 0.794394,
                "temperature": 0.9,
                "top_p": 1.0,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Tencent",
                "harness": null
            },
            "anthropic/claude-fable-5": {
                "accuracy": 68.82,
                "latency": 887.51,
                "stderr": 2.01,
                "cost_per_test": 11.594564,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.8-flash": {
                "accuracy": 68.483,
                "latency": 175.651,
                "stderr": 2.273,
                "cost_per_test": 1.995626,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "ant/ling-3.0-flash-af-rc3": {
                "accuracy": 67.853,
                "latency": 457.3,
                "stderr": 1.168,
                "cost_per_test": 0.059834,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "google/gemini-3.7-flash": {
                "accuracy": 67.744,
                "latency": 239.563,
                "stderr": 0.735,
                "cost_per_test": 1.503964,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.6-terra": {
                "accuracy": 66.945,
                "latency": 1555.117,
                "stderr": 6.337,
                "cost_per_test": 4.00508,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "meta/muse_spark_1_3": {
                "accuracy": 66.693,
                "latency": 326.775,
                "stderr": 1.04,
                "cost_per_test": 0.840906,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "xiaomi/mimo-v2.6-pro": {
                "accuracy": 66.658,
                "latency": 742.483,
                "stderr": 0.694,
                "cost_per_test": 0.244589,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 65.77,
                "latency": 764.736,
                "stderr": 1.75,
                "cost_per_test": 2.519775,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-4-8": {
                "accuracy": 65.706,
                "latency": 798.453,
                "stderr": 1.124,
                "cost_per_test": 4.989067,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "kimi/kimi-k3": {
                "accuracy": 65.052,
                "latency": 298.998,
                "stderr": 0.746,
                "cost_per_test": 2.326276,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "grok/grok-4.7": {
                "accuracy": 64.476,
                "latency": 993.887,
                "stderr": 0.513,
                "cost_per_test": 2.69861,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "grok/grok-4.6": {
                "accuracy": 64.355,
                "latency": 712.263,
                "stderr": 1.704,
                "cost_per_test": 1.548304,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-haiku-5-5": {
                "accuracy": 64.256,
                "latency": 2850.442,
                "stderr": 6.619,
                "cost_per_test": 2.670679,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.5-flash": {
                "accuracy": 63.948,
                "latency": 307.82,
                "stderr": 2.287,
                "cost_per_test": 2.588355,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "mistralai/mistral-large-4": {
                "accuracy": 63.866,
                "latency": 1799.166,
                "stderr": 2.745,
                "cost_per_test": 1.462122,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "meta/muse_spark_1_1": {
                "accuracy": 63.449,
                "latency": 569.177,
                "stderr": 1.377,
                "cost_per_test": 0.877309,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "openai/gpt-5.6-sol": {
                "accuracy": 63.329,
                "latency": 1405.72,
                "stderr": 3.028,
                "cost_per_test": 1.27232,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "deepseek/deepseek-v4-flash-0731": {
                "accuracy": 63.322,
                "latency": 600.738,
                "stderr": 1.754,
                "cost_per_test": 0.344028,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "openai/gpt-5.6-luna": {
                "accuracy": 62.836,
                "latency": 837.717,
                "stderr": 0.85,
                "cost_per_test": 0.317609,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-sonnet-5": {
                "accuracy": 62.076,
                "latency": 832.196,
                "stderr": 3.446,
                "cost_per_test": 3.696245,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.6-flash": {
                "accuracy": 61.968,
                "latency": 436.032,
                "stderr": 0.276,
                "cost_per_test": 1.347107,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-6-astra": {
                "accuracy": 61.684,
                "latency": 799.918,
                "stderr": 6.657,
                "cost_per_test": 8.137573,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-5.2": {
                "accuracy": 61.588,
                "latency": 372.664,
                "stderr": 1.686,
                "cost_per_test": 0.714998,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "anthropic/claude-opus-4-7": {
                "accuracy": 61.34,
                "latency": 364.799,
                "stderr": 0.27,
                "cost_per_test": 4.322777,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "deepseek/deepseek-v4.1-flash": {
                "accuracy": 61.312,
                "latency": 511.693,
                "stderr": 1.94,
                "cost_per_test": 0.281162,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "stepfun/step-5-preview": {
                "accuracy": 61.264,
                "latency": 835.787,
                "stderr": 0.77,
                "cost_per_test": 0.744177,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 1024000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Stepfun",
                "harness": null
            },
            "fireworks/ember-1": {
                "accuracy": 61.067,
                "latency": 885.57,
                "stderr": 6.604,
                "cost_per_test": 2.481802,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "alibaba/qwen3.8-max": {
                "accuracy": 60.549,
                "latency": 1545.939,
                "stderr": 2.357,
                "cost_per_test": 1.500045,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "alibaba/qwen3.8-27b": {
                "accuracy": 60.297,
                "latency": 777.945,
                "stderr": 6.595,
                "cost_per_test": 0.811063,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "deepseek/deepseek-v4-pro-0813": {
                "accuracy": 59.853,
                "latency": 1251.051,
                "stderr": 1.663,
                "cost_per_test": 1.11808,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "minimax/MiniMax-M3": {
                "accuracy": 59.831,
                "latency": 471.873,
                "stderr": 1.976,
                "cost_per_test": 0.346007,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 512000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "alibaba/qwen3.7-max": {
                "accuracy": 59.185,
                "latency": 694.575,
                "stderr": 1.751,
                "cost_per_test": 2.479458,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 58.909,
                "latency": 792.143,
                "stderr": 0.871,
                "cost_per_test": 4.884586,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-6.1-sol": {
                "accuracy": 58.406,
                "latency": 1045.456,
                "stderr": 0.616,
                "cost_per_test": 1.854069,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-6-luna": {
                "accuracy": 57.606,
                "latency": 1018.022,
                "stderr": 1.122,
                "cost_per_test": 0.147305,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4.5": {
                "accuracy": 56.412,
                "latency": 236.687,
                "stderr": 4.296,
                "cost_per_test": 1.05146,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-3.5-flash-lite": {
                "accuracy": 56.138,
                "latency": 227.605,
                "stderr": 1.176,
                "cost_per_test": 0.3482,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-6-sol": {
                "accuracy": 55.634,
                "latency": 560.943,
                "stderr": 1.007,
                "cost_per_test": 2.250652,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "thinkingmachines/inkling": {
                "accuracy": 53.24,
                "latency": 1007.923,
                "stderr": 3.425,
                "cost_per_test": 0.843749,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "xiaomi/mimo-v2.5-pro": {
                "accuracy": 53.148,
                "latency": 454.889,
                "stderr": 1.954,
                "cost_per_test": 0.249212,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "zai/glm-5.1": {
                "accuracy": 52.713,
                "latency": 955.188,
                "stderr": 2.288,
                "cost_per_test": 0.690792,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 52.649,
                "latency": 1666.135,
                "stderr": 2.117,
                "cost_per_test": 1.083494,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 52.124,
                "latency": 1512.854,
                "stderr": 2.522,
                "cost_per_test": 1.275044,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "nvidia/nemotron-3-ultra-550b-a55b": {
                "accuracy": 51.307,
                "latency": 250.558,
                "stderr": 0.192,
                "cost_per_test": null,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Nvidia",
                "harness": null
            },
            "thinkingmachines/inkling-small": {
                "accuracy": 50.754,
                "latency": 865.992,
                "stderr": 1.512,
                "cost_per_test": 0.238201,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 262000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 48.338,
                "latency": 2307.807,
                "stderr": 0.778,
                "cost_per_test": 1.372803,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4.3": {
                "accuracy": 47.688,
                "latency": 678.916,
                "stderr": 3.56,
                "cost_per_test": 0.896477,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 47.535,
                "latency": 666.694,
                "stderr": 6.673,
                "cost_per_test": 0.331588,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "xiaomi/mimo-v2.5": {
                "accuracy": 47.346,
                "latency": 331.617,
                "stderr": 4.019,
                "cost_per_test": 0.109825,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 46.181,
                "latency": 191.853,
                "stderr": 1.857,
                "cost_per_test": 1.695327,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 45.757,
                "latency": 255.838,
                "stderr": 3.161,
                "cost_per_test": 0.667368,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 44.547,
                "latency": 614.983,
                "stderr": 2.264,
                "cost_per_test": 0.572353,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "alibaba/qwen3.7-plus": {
                "accuracy": 44.441,
                "latency": 704.588,
                "stderr": 0.797,
                "cost_per_test": 0.364243,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 42.83,
                "latency": 356.213,
                "stderr": 3.444,
                "cost_per_test": 0.183174,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 42.517,
                "latency": 192.263,
                "stderr": 1.158,
                "cost_per_test": 0.782271,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "ant/ling-3.0-flash-2607": {
                "accuracy": 41.945,
                "latency": 182.286,
                "stderr": 3.762,
                "cost_per_test": 0.064081,
                "temperature": 0.6,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "mistralai/mistral-medium-3.5": {
                "accuracy": 40.223,
                "latency": 929.601,
                "stderr": 1.634,
                "cost_per_test": 1.750126,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 39.954,
                "latency": 347.25,
                "stderr": 0.851,
                "cost_per_test": 0.833034,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 37.38,
                "latency": 54.432,
                "stderr": 2.128,
                "cost_per_test": 0.162369,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 34.96,
                "latency": 680.199,
                "stderr": 1.681,
                "cost_per_test": 0.189858,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "poolside/laguna-m.1": {
                "accuracy": 30.507,
                "latency": 1019.588,
                "stderr": 4.407,
                "cost_per_test": 0.594924,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "fireworks/nemotron-lightning-3p5-30b-a3b": {
                "accuracy": 20.571,
                "latency": 314.357,
                "stderr": 1.801,
                "cost_per_test": 0.032094,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "inception/mercury-2.5": {
                "accuracy": 19.063,
                "latency": 119.47,
                "stderr": 1.364,
                "cost_per_test": 0.111069,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Inception",
                "harness": null
            },
            "poolside/laguna-xs.2": {
                "accuracy": 18.734,
                "latency": 481.877,
                "stderr": 1.143,
                "cost_per_test": 0.163627,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "cohere/command-a-plus-05-2026": {
                "accuracy": 7.584,
                "latency": 581.011,
                "stderr": 0.695,
                "cost_per_test": 1.512103,
                "temperature": 0.9,
                "top_p": 0.95,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            }
        },
        "earnings_analysis": {
            "google/gemini-4-argon": {
                "accuracy": 84.807,
                "latency": 721.062,
                "stderr": 0.667,
                "cost_per_test": 2.470687,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.8-flash": {
                "accuracy": 81.39,
                "latency": 170.953,
                "stderr": 2.043,
                "cost_per_test": 1.167384,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.7-flash": {
                "accuracy": 79.057,
                "latency": 144.716,
                "stderr": 1.32,
                "cost_per_test": 0.870455,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.5-flash": {
                "accuracy": 79.013,
                "latency": 243.066,
                "stderr": 1.341,
                "cost_per_test": 1.266639,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "zai/glm-5.3": {
                "accuracy": 78.692,
                "latency": 763.265,
                "stderr": 5.675,
                "cost_per_test": 0.470444,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "anthropic/claude-opus-5-5": {
                "accuracy": 77.992,
                "latency": 1320.967,
                "stderr": 0.633,
                "cost_per_test": 4.436471,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "xiaomi/mimo-v2.6-pro": {
                "accuracy": 77.669,
                "latency": 435.172,
                "stderr": 1.519,
                "cost_per_test": 0.059756,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "meta/muse_spark_1_2": {
                "accuracy": 77.298,
                "latency": 251.461,
                "stderr": 2.401,
                "cost_per_test": 0.363992,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "anthropic/claude-fable-5": {
                "accuracy": 77.263,
                "latency": 369.254,
                "stderr": 1.276,
                "cost_per_test": 2.990129,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.6-terra": {
                "accuracy": 77.263,
                "latency": 760.584,
                "stderr": 5.889,
                "cost_per_test": 1.418824,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "meta/muse_spark_1_3_max": {
                "accuracy": 77.158,
                "latency": 118.889,
                "stderr": 5.885,
                "cost_per_test": 0.288063,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "meta/muse_spark_1_3": {
                "accuracy": 77.088,
                "latency": 204.882,
                "stderr": 2.37,
                "cost_per_test": 0.29674,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "anthropic/claude-sonnet-5-5": {
                "accuracy": 76.522,
                "latency": 1007.55,
                "stderr": 1.678,
                "cost_per_test": 2.437318,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "mistralai/mistral-large-4": {
                "accuracy": 76.491,
                "latency": 1055.885,
                "stderr": 0.669,
                "cost_per_test": 0.485529,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "meta/muse_spark_1_1": {
                "accuracy": 76.197,
                "latency": 240.689,
                "stderr": 0.67,
                "cost_per_test": 0.282783,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "anthropic/claude-fable-5-1": {
                "accuracy": 75.474,
                "latency": 449.038,
                "stderr": 6.082,
                "cost_per_test": 2.711131,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.6-flash": {
                "accuracy": 75.285,
                "latency": 392.353,
                "stderr": 1.084,
                "cost_per_test": 0.835789,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "tencent/hy4-preview": {
                "accuracy": 75.053,
                "latency": 1176.754,
                "stderr": 2.037,
                "cost_per_test": 0.257835,
                "temperature": 0.9,
                "top_p": 1.0,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Tencent",
                "harness": null
            },
            "ant/ling-3.0-flash-af-rc3": {
                "accuracy": 74.943,
                "latency": 225.815,
                "stderr": 1.308,
                "cost_per_test": 0.016179,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "openai/gpt-5.6-luna": {
                "accuracy": 74.875,
                "latency": 390.397,
                "stderr": 1.278,
                "cost_per_test": 0.109633,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4.7": {
                "accuracy": 74.585,
                "latency": 576.759,
                "stderr": 0.556,
                "cost_per_test": 0.964679,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "xiaomi/mimo-v2.6-flash": {
                "accuracy": 74.518,
                "latency": 259.923,
                "stderr": 0.931,
                "cost_per_test": 0.025612,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "openai/gpt-5.6-sol": {
                "accuracy": 74.351,
                "latency": 423.559,
                "stderr": 1.294,
                "cost_per_test": 0.687883,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-6-luna": {
                "accuracy": 74.193,
                "latency": 386.01,
                "stderr": 0.742,
                "cost_per_test": 0.040595,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4.6": {
                "accuracy": 73.86,
                "latency": 389.506,
                "stderr": 0.551,
                "cost_per_test": 0.572553,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-opus-5": {
                "accuracy": 73.789,
                "latency": 322.978,
                "stderr": 0.844,
                "cost_per_test": 1.574326,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-haiku-5-5": {
                "accuracy": 73.474,
                "latency": 1079.651,
                "stderr": 6.244,
                "cost_per_test": 0.311867,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 73.158,
                "latency": 381.734,
                "stderr": 0.0,
                "cost_per_test": 1.633725,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-5.3-flash": {
                "accuracy": 72.902,
                "latency": 1041.512,
                "stderr": 6.22,
                "cost_per_test": 0.017052,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "deepseek/deepseek-v4.1-flash": {
                "accuracy": 72.881,
                "latency": 244.529,
                "stderr": 2.846,
                "cost_per_test": 0.080051,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "anthropic/claude-sonnet-5": {
                "accuracy": 72.588,
                "latency": 434.462,
                "stderr": 0.589,
                "cost_per_test": 0.714731,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-6-sol": {
                "accuracy": 72.38,
                "latency": 254.621,
                "stderr": 0.616,
                "cost_per_test": 0.639986,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-4-8": {
                "accuracy": 72.07,
                "latency": 444.246,
                "stderr": 1.3,
                "cost_per_test": 1.386113,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "stepfun/step-5-preview": {
                "accuracy": 71.702,
                "latency": 600.416,
                "stderr": 0.882,
                "cost_per_test": 0.222031,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 1024000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Stepfun",
                "harness": null
            },
            "anthropic/claude-opus-4-7": {
                "accuracy": 71.562,
                "latency": 264.962,
                "stderr": 1.312,
                "cost_per_test": 1.342081,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-6.1-sol": {
                "accuracy": 71.474,
                "latency": 387.076,
                "stderr": 0.0,
                "cost_per_test": 0.462746,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3.8-27b": {
                "accuracy": 70.936,
                "latency": 453.466,
                "stderr": 6.351,
                "cost_per_test": 0.450356,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "grok/grok-4.5": {
                "accuracy": 70.848,
                "latency": 178.595,
                "stderr": 1.094,
                "cost_per_test": 0.373969,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "deepseek/deepseek-v4-flash-0731": {
                "accuracy": 69.777,
                "latency": 263.446,
                "stderr": 1.773,
                "cost_per_test": 0.087508,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "openai/gpt-6-astra": {
                "accuracy": 69.474,
                "latency": 303.329,
                "stderr": 6.518,
                "cost_per_test": 2.285633,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "kimi/kimi-k3": {
                "accuracy": 69.418,
                "latency": 119.68,
                "stderr": 1.089,
                "cost_per_test": 0.609811,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "alibaba/qwen3.8-max": {
                "accuracy": 67.734,
                "latency": 951.983,
                "stderr": 2.763,
                "cost_per_test": 0.549896,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "fireworks/ember-1": {
                "accuracy": 66.947,
                "latency": 346.511,
                "stderr": 6.608,
                "cost_per_test": 0.590222,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "zai/glm-5.2": {
                "accuracy": 66.596,
                "latency": 289.351,
                "stderr": 2.878,
                "cost_per_test": 0.286337,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "deepseek/deepseek-v4-pro-0813": {
                "accuracy": 66.463,
                "latency": 635.73,
                "stderr": 3.249,
                "cost_per_test": 0.327246,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "minimax/MiniMax-M3": {
                "accuracy": 66.191,
                "latency": 303.41,
                "stderr": 1.409,
                "cost_per_test": 0.130797,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 512000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 65.294,
                "latency": 1636.237,
                "stderr": 0.426,
                "cost_per_test": 0.754578,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 65.119,
                "latency": 427.22,
                "stderr": 2.993,
                "cost_per_test": 1.05356,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 64.654,
                "latency": 154.84,
                "stderr": 0.648,
                "cost_per_test": 0.509894,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.5-flash-lite": {
                "accuracy": 64.369,
                "latency": 381.425,
                "stderr": 1.232,
                "cost_per_test": 0.230543,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 63.824,
                "latency": 847.882,
                "stderr": 0.746,
                "cost_per_test": 0.491388,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 62.076,
                "latency": 193.438,
                "stderr": 0.589,
                "cost_per_test": 0.27137,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "thinkingmachines/inkling": {
                "accuracy": 60.359,
                "latency": 773.178,
                "stderr": 1.955,
                "cost_per_test": 0.437074,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "alibaba/qwen3.7-max": {
                "accuracy": 60.164,
                "latency": 488.6,
                "stderr": 1.907,
                "cost_per_test": 1.364611,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 59.84,
                "latency": 1036.974,
                "stderr": 1.503,
                "cost_per_test": 0.320383,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "zai/glm-5.1": {
                "accuracy": 56.958,
                "latency": 526.957,
                "stderr": 0.613,
                "cost_per_test": 0.319196,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "thinkingmachines/inkling-small": {
                "accuracy": 55.383,
                "latency": 485.224,
                "stderr": 1.457,
                "cost_per_test": 0.130394,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 262000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "xiaomi/mimo-v2.5-pro": {
                "accuracy": 54.614,
                "latency": 381.5,
                "stderr": 4.256,
                "cost_per_test": 0.066769,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 52.398,
                "latency": 440.575,
                "stderr": 0.998,
                "cost_per_test": 0.344742,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "alibaba/qwen3.7-plus": {
                "accuracy": 50.339,
                "latency": 616.533,
                "stderr": 2.073,
                "cost_per_test": 0.143864,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "grok/grok-4.3": {
                "accuracy": 47.734,
                "latency": 292.659,
                "stderr": 3.601,
                "cost_per_test": 0.335047,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 45.604,
                "latency": 239.547,
                "stderr": 1.629,
                "cost_per_test": 0.07704,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "xiaomi/mimo-v2.5": {
                "accuracy": 45.006,
                "latency": 269.15,
                "stderr": 1.467,
                "cost_per_test": 0.030265,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "mistralai/mistral-medium-3.5": {
                "accuracy": 42.125,
                "latency": 548.402,
                "stderr": 2.287,
                "cost_per_test": 1.725068,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "nvidia/nemotron-3-ultra-550b-a55b": {
                "accuracy": 40.623,
                "latency": 203.776,
                "stderr": 1.887,
                "cost_per_test": null,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Nvidia",
                "harness": null
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 37.386,
                "latency": 463.954,
                "stderr": 6.709,
                "cost_per_test": 0.12981,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "ant/ling-3.0-flash-2607": {
                "accuracy": 37.0,
                "latency": 126.343,
                "stderr": 2.199,
                "cost_per_test": 0.035294,
                "temperature": 0.6,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 34.433,
                "latency": 166.506,
                "stderr": 1.683,
                "cost_per_test": 0.297602,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 31.848,
                "latency": 499.749,
                "stderr": 1.171,
                "cost_per_test": 0.097748,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 31.261,
                "latency": 57.997,
                "stderr": 0.373,
                "cost_per_test": 0.08654,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 26.692,
                "latency": 110.802,
                "stderr": 1.753,
                "cost_per_test": 0.305043,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "poolside/laguna-m.1": {
                "accuracy": 25.937,
                "latency": 719.33,
                "stderr": 2.513,
                "cost_per_test": 0.511059,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "fireworks/nemotron-lightning-3p5-30b-a3b": {
                "accuracy": 15.59,
                "latency": 561.93,
                "stderr": 2.232,
                "cost_per_test": 0.019569,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "inception/mercury-2.5": {
                "accuracy": 15.394,
                "latency": 104.1,
                "stderr": 1.754,
                "cost_per_test": 0.074738,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Inception",
                "harness": null
            },
            "poolside/laguna-xs.2": {
                "accuracy": 15.297,
                "latency": 558.171,
                "stderr": 2.628,
                "cost_per_test": 0.193247,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "cohere/command-a-plus-05-2026": {
                "accuracy": 7.963,
                "latency": 568.404,
                "stderr": 2.364,
                "cost_per_test": 1.368424,
                "temperature": 0.9,
                "top_p": 0.95,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            }
        },
        "financial_modeling": {
            "meta/muse_spark_1_2": {
                "accuracy": 34.522,
                "latency": 355.959,
                "stderr": 2.161,
                "cost_per_test": 0.743758,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "meta/muse_spark_1_1": {
                "accuracy": 34.139,
                "latency": 380.427,
                "stderr": 2.315,
                "cost_per_test": 0.668285,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "meta/muse_spark_1_3": {
                "accuracy": 33.512,
                "latency": 375.168,
                "stderr": 2.891,
                "cost_per_test": 0.74606,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "google/gemini-3.8-flash": {
                "accuracy": 32.646,
                "latency": 256.536,
                "stderr": 0.605,
                "cost_per_test": 2.172068,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-4-argon": {
                "accuracy": 29.803,
                "latency": 923.889,
                "stderr": 1.815,
                "cost_per_test": 3.869962,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.5-flash": {
                "accuracy": 29.372,
                "latency": 424.764,
                "stderr": 0.936,
                "cost_per_test": 2.97355,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "meta/muse_spark_1_3_max": {
                "accuracy": 28.652,
                "latency": 262.662,
                "stderr": 6.088,
                "cost_per_test": 0.790029,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "anthropic/claude-haiku-5-5": {
                "accuracy": 28.33,
                "latency": 1096.71,
                "stderr": 6.039,
                "cost_per_test": 1.081013,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-5-5": {
                "accuracy": 26.886,
                "latency": 1192.947,
                "stderr": 1.186,
                "cost_per_test": 6.369203,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-5": {
                "accuracy": 26.65,
                "latency": 616.7,
                "stderr": 0.863,
                "cost_per_test": 3.702148,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-sonnet-5-5": {
                "accuracy": 25.507,
                "latency": 1145.389,
                "stderr": 1.111,
                "cost_per_test": 4.340797,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "zai/glm-5.3-flash": {
                "accuracy": 25.4,
                "latency": 704.527,
                "stderr": 5.947,
                "cost_per_test": 0.048996,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "anthropic/claude-fable-5-1": {
                "accuracy": 25.326,
                "latency": 624.838,
                "stderr": 5.934,
                "cost_per_test": 6.89102,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.6-flash": {
                "accuracy": 25.122,
                "latency": 531.046,
                "stderr": 0.605,
                "cost_per_test": 1.610789,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "xiaomi/mimo-v2.6-pro": {
                "accuracy": 24.83,
                "latency": 1022.455,
                "stderr": 1.679,
                "cost_per_test": 0.189253,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "xiaomi/mimo-v2.6-flash": {
                "accuracy": 24.202,
                "latency": 614.468,
                "stderr": 1.67,
                "cost_per_test": 0.065996,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "anthropic/claude-sonnet-5": {
                "accuracy": 24.197,
                "latency": 709.547,
                "stderr": 1.167,
                "cost_per_test": 2.30529,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.7-flash": {
                "accuracy": 24.155,
                "latency": 210.797,
                "stderr": 0.327,
                "cost_per_test": 1.602877,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "kimi/kimi-k3": {
                "accuracy": 23.556,
                "latency": 244.743,
                "stderr": 0.991,
                "cost_per_test": 1.672816,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "anthropic/claude-opus-4-8": {
                "accuracy": 23.347,
                "latency": 669.744,
                "stderr": 1.029,
                "cost_per_test": 4.205596,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "fireworks/ember-1": {
                "accuracy": 23.163,
                "latency": 497.068,
                "stderr": 5.673,
                "cost_per_test": 1.437298,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "openai/gpt-6.1-sol": {
                "accuracy": 23.093,
                "latency": 861.831,
                "stderr": 1.237,
                "cost_per_test": 1.30253,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "deepseek/deepseek-v4.1-flash": {
                "accuracy": 23.052,
                "latency": 412.644,
                "stderr": 2.835,
                "cost_per_test": 0.192114,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "openai/gpt-5.6-terra": {
                "accuracy": 23.031,
                "latency": 1130.479,
                "stderr": 5.659,
                "cost_per_test": 2.768448,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 22.93,
                "latency": 592.855,
                "stderr": 2.11,
                "cost_per_test": 3.178322,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-6-luna": {
                "accuracy": 22.493,
                "latency": 900.079,
                "stderr": 1.533,
                "cost_per_test": 0.106309,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-5.3": {
                "accuracy": 22.05,
                "latency": 886.228,
                "stderr": 5.689,
                "cost_per_test": 0.955818,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "openai/gpt-5.6-luna": {
                "accuracy": 21.59,
                "latency": 707.471,
                "stderr": 1.6,
                "cost_per_test": 0.232812,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-6-astra": {
                "accuracy": 21.585,
                "latency": 586.329,
                "stderr": 5.585,
                "cost_per_test": 5.220567,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "tencent/hy4-preview": {
                "accuracy": 20.887,
                "latency": 1085.17,
                "stderr": 0.663,
                "cost_per_test": 0.444398,
                "temperature": 0.9,
                "top_p": 1.0,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Tencent",
                "harness": null
            },
            "openai/gpt-5.6-sol": {
                "accuracy": 20.706,
                "latency": 968.534,
                "stderr": 0.501,
                "cost_per_test": 1.476372,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "mistralai/mistral-large-4": {
                "accuracy": 20.456,
                "latency": 1830.614,
                "stderr": 1.548,
                "cost_per_test": 1.315313,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "alibaba/qwen3.8-max": {
                "accuracy": 20.389,
                "latency": 1108.071,
                "stderr": 2.448,
                "cost_per_test": 1.113657,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "anthropic/claude-opus-4-7": {
                "accuracy": 20.096,
                "latency": 505.87,
                "stderr": 1.765,
                "cost_per_test": 4.408222,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "deepseek/deepseek-v4-flash-0731": {
                "accuracy": 19.84,
                "latency": 392.596,
                "stderr": 1.171,
                "cost_per_test": 0.231681,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 19.617,
                "latency": 2159.906,
                "stderr": 1.636,
                "cost_per_test": 1.074981,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3.5-flash-lite": {
                "accuracy": 19.323,
                "latency": 284.147,
                "stderr": 1.496,
                "cost_per_test": 0.497843,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-6-sol": {
                "accuracy": 19.298,
                "latency": 485.214,
                "stderr": 0.93,
                "cost_per_test": 1.624618,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3.8-27b": {
                "accuracy": 19.297,
                "latency": 841.903,
                "stderr": 5.278,
                "cost_per_test": 0.740873,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "grok/grok-4.6": {
                "accuracy": 18.906,
                "latency": 1780.818,
                "stderr": 2.335,
                "cost_per_test": 2.407263,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "grok/grok-4.7": {
                "accuracy": 17.66,
                "latency": 1103.214,
                "stderr": 1.255,
                "cost_per_test": 2.887908,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-fable-5": {
                "accuracy": 17.609,
                "latency": 599.956,
                "stderr": 1.578,
                "cost_per_test": 7.073379,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 17.561,
                "latency": 942.334,
                "stderr": 1.726,
                "cost_per_test": 2.769029,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3.7-max": {
                "accuracy": 16.961,
                "latency": 968.957,
                "stderr": 2.341,
                "cost_per_test": 2.284393,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "deepseek/deepseek-v4-pro-0813": {
                "accuracy": 16.836,
                "latency": 973.913,
                "stderr": 1.475,
                "cost_per_test": 0.766127,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 16.728,
                "latency": 762.437,
                "stderr": 1.148,
                "cost_per_test": 0.961778,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "stepfun/step-5-preview": {
                "accuracy": 16.714,
                "latency": 703.989,
                "stderr": 1.524,
                "cost_per_test": 0.569585,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 1024000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Stepfun",
                "harness": null
            },
            "thinkingmachines/inkling": {
                "accuracy": 16.706,
                "latency": 1625.277,
                "stderr": 1.673,
                "cost_per_test": 1.276114,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "ant/ling-3.0-flash-af-rc3": {
                "accuracy": 16.266,
                "latency": 295.882,
                "stderr": 1.704,
                "cost_per_test": 0.039193,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "grok/grok-4.5": {
                "accuracy": 15.619,
                "latency": 471.26,
                "stderr": 1.775,
                "cost_per_test": 1.54595,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 15.101,
                "latency": 287.693,
                "stderr": 1.726,
                "cost_per_test": 1.714516,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 14.434,
                "latency": 250.719,
                "stderr": 2.066,
                "cost_per_test": 0.486996,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 14.201,
                "latency": 1736.611,
                "stderr": 1.558,
                "cost_per_test": 1.336633,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "minimax/MiniMax-M3": {
                "accuracy": 13.99,
                "latency": 674.781,
                "stderr": 2.011,
                "cost_per_test": 0.358739,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 512000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 13.49,
                "latency": 368.43,
                "stderr": 1.455,
                "cost_per_test": 0.125942,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-5.1": {
                "accuracy": 13.466,
                "latency": 1345.298,
                "stderr": 1.012,
                "cost_per_test": 0.848763,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "zai/glm-5.2": {
                "accuracy": 12.885,
                "latency": 777.094,
                "stderr": 1.787,
                "cost_per_test": 0.926811,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 10.548,
                "latency": 1617.918,
                "stderr": 2.633,
                "cost_per_test": 0.803012,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "thinkingmachines/inkling-small": {
                "accuracy": 10.316,
                "latency": 1069.12,
                "stderr": 1.563,
                "cost_per_test": 0.288464,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 262000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "alibaba/qwen3.7-plus": {
                "accuracy": 10.057,
                "latency": 802.253,
                "stderr": 0.985,
                "cost_per_test": 0.361189,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "xiaomi/mimo-v2.5-pro": {
                "accuracy": 9.876,
                "latency": 786.146,
                "stderr": 1.368,
                "cost_per_test": 0.204645,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "nvidia/nemotron-3-ultra-550b-a55b": {
                "accuracy": 7.071,
                "latency": 312.219,
                "stderr": 1.766,
                "cost_per_test": null,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Nvidia",
                "harness": null
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 6.582,
                "latency": 775.011,
                "stderr": 2.968,
                "cost_per_test": 0.362877,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "grok/grok-4.3": {
                "accuracy": 5.914,
                "latency": 657.335,
                "stderr": 0.201,
                "cost_per_test": 1.036963,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "ant/ling-3.0-flash-2607": {
                "accuracy": 5.307,
                "latency": 205.99,
                "stderr": 1.043,
                "cost_per_test": 0.055804,
                "temperature": 0.6,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "xiaomi/mimo-v2.5": {
                "accuracy": 5.271,
                "latency": 594.806,
                "stderr": 0.598,
                "cost_per_test": 0.091582,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 5.141,
                "latency": 214.099,
                "stderr": 1.119,
                "cost_per_test": 0.652418,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "mistralai/mistral-medium-3.5": {
                "accuracy": 5.141,
                "latency": 907.652,
                "stderr": 0.821,
                "cost_per_test": 2.387633,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "fireworks/nemotron-lightning-3p5-30b-a3b": {
                "accuracy": 4.604,
                "latency": 320.226,
                "stderr": 0.074,
                "cost_per_test": 0.034263,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 4.112,
                "latency": 60.912,
                "stderr": 1.206,
                "cost_per_test": 0.141152,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 4.073,
                "latency": 1482.909,
                "stderr": 0.778,
                "cost_per_test": 0.308759,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "inception/mercury-2.5": {
                "accuracy": 2.775,
                "latency": 113.172,
                "stderr": 0.085,
                "cost_per_test": 0.098027,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Inception",
                "harness": null
            },
            "poolside/laguna-m.1": {
                "accuracy": 2.611,
                "latency": 1755.714,
                "stderr": 0.944,
                "cost_per_test": 1.934058,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "poolside/laguna-xs.2": {
                "accuracy": 2.591,
                "latency": 819.584,
                "stderr": 0.591,
                "cost_per_test": 0.327328,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 2.447,
                "latency": 387.791,
                "stderr": 0.959,
                "cost_per_test": 0.905306,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "cohere/command-a-plus-05-2026": {
                "accuracy": 0.763,
                "latency": 506.787,
                "stderr": 0.144,
                "cost_per_test": 2.31425,
                "temperature": 0.9,
                "top_p": 0.95,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            }
        },
        "all_pass": {
            "google/gemini-4-argon": {
                "accuracy": 55.789,
                "latency": 980.524,
                "stderr": 0.265,
                "cost_per_test": 4.378161,
                "token_totals": {
                    "input_tokens": 609679241,
                    "output_tokens": 31503850,
                    "reasoning_tokens": 13402361,
                    "cache_read_tokens": 289110948,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "meta/muse_spark_1_2": {
                "accuracy": 50.881,
                "latency": 304.205,
                "stderr": 0.606,
                "cost_per_test": 0.76972,
                "token_totals": {
                    "input_tokens": 424942458,
                    "output_tokens": 12218400,
                    "reasoning_tokens": 6559929,
                    "cache_read_tokens": null,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "google/gemini-3.8-flash": {
                "accuracy": 49.694,
                "latency": 200.952,
                "stderr": 0.424,
                "cost_per_test": 2.000457,
                "token_totals": {
                    "input_tokens": 1118010382,
                    "output_tokens": 25909358,
                    "reasoning_tokens": 18104089,
                    "cache_read_tokens": 719355707,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "meta/muse_spark_1_3_max": {
                "accuracy": 49.514,
                "latency": 198.929,
                "stderr": 2.188,
                "cost_per_test": 0.756512,
                "token_totals": {
                    "input_tokens": 508981355,
                    "output_tokens": 16233804,
                    "reasoning_tokens": 9783963,
                    "cache_read_tokens": 331627422,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "meta/muse_spark_1_3": {
                "accuracy": 48.737,
                "latency": 341.936,
                "stderr": 0.417,
                "cost_per_test": 0.741132,
                "token_totals": {
                    "input_tokens": 413247670,
                    "output_tokens": 14548497,
                    "reasoning_tokens": 8843980,
                    "cache_read_tokens": null,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "anthropic/claude-opus-5-5": {
                "accuracy": 48.111,
                "latency": 1987.81,
                "stderr": 0.622,
                "cost_per_test": 9.215734,
                "token_totals": {
                    "input_tokens": 763751733,
                    "output_tokens": 109304455,
                    "reasoning_tokens": 94811735,
                    "cache_read_tokens": 413590027,
                    "cache_write_tokens": 350117722
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-fable-5-1": {
                "accuracy": 47.773,
                "latency": 1094.165,
                "stderr": 2.195,
                "cost_per_test": 8.346696,
                "token_totals": {
                    "input_tokens": 458174812,
                    "output_tokens": 33030168,
                    "reasoning_tokens": 19377315,
                    "cache_read_tokens": 228829928,
                    "cache_write_tokens": 229316981
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-5": {
                "accuracy": 47.708,
                "latency": 598.424,
                "stderr": 0.201,
                "cost_per_test": 5.124473,
                "token_totals": {
                    "input_tokens": 543793980,
                    "output_tokens": 22921224,
                    "reasoning_tokens": 9150666,
                    "cache_read_tokens": 253099566,
                    "cache_write_tokens": 290670709
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-sonnet-5-5": {
                "accuracy": 47.59,
                "latency": 1903.336,
                "stderr": 0.147,
                "cost_per_test": 6.552431,
                "token_totals": {
                    "input_tokens": 1121526513,
                    "output_tokens": 137870326,
                    "reasoning_tokens": 120529788,
                    "cache_read_tokens": 555448124,
                    "cache_write_tokens": 566027579
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.7-flash": {
                "accuracy": 47.45,
                "latency": 181.334,
                "stderr": 0.2,
                "cost_per_test": 1.477884,
                "token_totals": {
                    "input_tokens": 711083728,
                    "output_tokens": 16887053,
                    "reasoning_tokens": 10058516,
                    "cache_read_tokens": 391282104,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "zai/glm-5.3-flash": {
                "accuracy": 46.259,
                "latency": 890.519,
                "stderr": 2.188,
                "cost_per_test": 0.047625,
                "token_totals": {
                    "input_tokens": 333663746,
                    "output_tokens": 26908803,
                    "reasoning_tokens": null,
                    "cache_read_tokens": 172014080,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "xiaomi/mimo-v2.6-pro": {
                "accuracy": 45.837,
                "latency": 648.075,
                "stderr": 0.199,
                "cost_per_test": 0.198909,
                "token_totals": {
                    "input_tokens": 305876104,
                    "output_tokens": 13378019,
                    "reasoning_tokens": 7659784,
                    "cache_read_tokens": 127922645,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "google/gemini-3.5-flash": {
                "accuracy": 45.706,
                "latency": 322.279,
                "stderr": 0.196,
                "cost_per_test": 2.505488,
                "token_totals": {
                    "input_tokens": 1227406624,
                    "output_tokens": 26782944,
                    "reasoning_tokens": 20834822,
                    "cache_read_tokens": 707613227,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-fable-5": {
                "accuracy": 45.694,
                "latency": 610.952,
                "stderr": 0.871,
                "cost_per_test": 8.061563,
                "token_totals": {
                    "input_tokens": 311882539,
                    "output_tokens": 19969523,
                    "reasoning_tokens": null,
                    "cache_read_tokens": 110357130,
                    "cache_write_tokens": 201446408
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "meta/muse_spark_1_1": {
                "accuracy": 44.919,
                "latency": 414.77,
                "stderr": 1.004,
                "cost_per_test": 0.723391,
                "token_totals": {
                    "input_tokens": 363982665,
                    "output_tokens": 15279252,
                    "reasoning_tokens": 10574096,
                    "cache_read_tokens": 177594427,
                    "cache_write_tokens": null
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "zai/glm-5.3": {
                "accuracy": 44.874,
                "latency": 949.539,
                "stderr": 2.17,
                "cost_per_test": 1.072768,
                "token_totals": {
                    "input_tokens": 363886559,
                    "output_tokens": 36729670,
                    "reasoning_tokens": null,
                    "cache_read_tokens": null,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "google/gemini-3.6-flash": {
                "accuracy": 44.634,
                "latency": 224.899,
                "stderr": 0.225,
                "cost_per_test": 1.402901,
                "token_totals": {
                    "input_tokens": 627928399,
                    "output_tokens": 16297009,
                    "reasoning_tokens": 11292206,
                    "cache_read_tokens": 320603458,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "xiaomi/mimo-v2.6-flash": {
                "accuracy": 44.427,
                "latency": 421.99,
                "stderr": 0.281,
                "cost_per_test": 0.071536,
                "token_totals": {
                    "input_tokens": 349885254,
                    "output_tokens": 15078803,
                    "reasoning_tokens": 7475980,
                    "cache_read_tokens": 154212352,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "tencent/hy4-preview": {
                "accuracy": 44.017,
                "latency": 1424.632,
                "stderr": 0.231,
                "cost_per_test": 0.5903,
                "token_totals": {
                    "input_tokens": 316560636,
                    "output_tokens": 41494337,
                    "reasoning_tokens": 34844249,
                    "cache_read_tokens": 129727253,
                    "cache_write_tokens": 0
                },
                "temperature": 0.9,
                "top_p": 1.0,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Tencent",
                "harness": null
            },
            "mistralai/mistral-large-4": {
                "accuracy": 43.787,
                "latency": 1561.879,
                "stderr": 0.164,
                "cost_per_test": 1.201399,
                "token_totals": {
                    "input_tokens": 365895891,
                    "output_tokens": 27874175,
                    "reasoning_tokens": null,
                    "cache_read_tokens": 60248405,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "anthropic/claude-opus-4-8": {
                "accuracy": 43.517,
                "latency": 547.397,
                "stderr": 0.303,
                "cost_per_test": 4.218538,
                "token_totals": {
                    "input_tokens": 309993285,
                    "output_tokens": 17301258,
                    "reasoning_tokens": null,
                    "cache_read_tokens": 82013466,
                    "cache_write_tokens": 227923817
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "ant/ling-3.0-flash-af-rc3": {
                "accuracy": 43.361,
                "latency": 375.066,
                "stderr": 0.709,
                "cost_per_test": 0.044792,
                "token_totals": {
                    "input_tokens": 401804903,
                    "output_tokens": 19410118,
                    "reasoning_tokens": null,
                    "cache_read_tokens": 155120896,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "openai/gpt-5.6-luna": {
                "accuracy": 43.285,
                "latency": 772.153,
                "stderr": 0.047,
                "cost_per_test": 0.279224,
                "token_totals": {
                    "input_tokens": 533549683,
                    "output_tokens": 52574268,
                    "reasoning_tokens": 45760878,
                    "cache_read_tokens": null,
                    "cache_write_tokens": null
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4.6": {
                "accuracy": 42.901,
                "latency": 1000.937,
                "stderr": 0.911,
                "cost_per_test": 1.657732,
                "token_totals": {
                    "input_tokens": 495041785,
                    "output_tokens": 24551729,
                    "reasoning_tokens": 16279727,
                    "cache_read_tokens": 260943147,
                    "cache_write_tokens": null
                },
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-haiku-5-5": {
                "accuracy": 42.874,
                "latency": 1849.0,
                "stderr": 2.184,
                "cost_per_test": 1.636603,
                "token_totals": {
                    "input_tokens": 1125304863,
                    "output_tokens": 207972605,
                    "reasoning_tokens": 195320438,
                    "cache_read_tokens": 434937131,
                    "cache_write_tokens": 690320292
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.6-terra": {
                "accuracy": 42.756,
                "latency": 1449.852,
                "stderr": 2.158,
                "cost_per_test": 3.629197,
                "token_totals": {
                    "input_tokens": 651703808,
                    "output_tokens": 67770045,
                    "reasoning_tokens": 58991995,
                    "cache_read_tokens": 334920434,
                    "cache_write_tokens": null
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4.7": {
                "accuracy": 42.137,
                "latency": 932.724,
                "stderr": 0.348,
                "cost_per_test": 2.719136,
                "token_totals": {
                    "input_tokens": 827415903,
                    "output_tokens": 23428536,
                    "reasoning_tokens": 15352966,
                    "cache_read_tokens": 444856320,
                    "cache_write_tokens": null
                },
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-sonnet-5": {
                "accuracy": 42.084,
                "latency": 792.086,
                "stderr": 0.357,
                "cost_per_test": 0.750759,
                "token_totals": {
                    "input_tokens": 164610121,
                    "output_tokens": 10102005,
                    "reasoning_tokens": null,
                    "cache_read_tokens": 75956629,
                    "cache_write_tokens": 88646581
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "kimi/kimi-k3": {
                "accuracy": 41.837,
                "latency": 285.753,
                "stderr": 0.281,
                "cost_per_test": 1.913224,
                "token_totals": {
                    "input_tokens": 460080913,
                    "output_tokens": 16314784,
                    "reasoning_tokens": 10722175,
                    "cache_read_tokens": 282968006,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "openai/gpt-5.6-sol": {
                "accuracy": 41.362,
                "latency": 1165.258,
                "stderr": 0.068,
                "cost_per_test": 1.253419,
                "token_totals": {
                    "input_tokens": 322058918,
                    "output_tokens": 6216567,
                    "reasoning_tokens": 4178719,
                    "cache_read_tokens": 276042752,
                    "cache_write_tokens": null
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "deepseek/deepseek-v4.1-flash": {
                "accuracy": 41.052,
                "latency": 364.376,
                "stderr": 0.446,
                "cost_per_test": 0.207465,
                "token_totals": {
                    "input_tokens": 354713236,
                    "output_tokens": 33772547,
                    "reasoning_tokens": 28648435,
                    "cache_read_tokens": 182250763,
                    "cache_write_tokens": null
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "openai/gpt-6-astra": {
                "accuracy": 40.252,
                "latency": 692.672,
                "stderr": 2.115,
                "cost_per_test": 6.81746,
                "token_totals": {
                    "input_tokens": 240568066,
                    "output_tokens": 22397666,
                    "reasoning_tokens": 16813955,
                    "cache_read_tokens": 70591920,
                    "cache_write_tokens": null
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-6.1-sol": {
                "accuracy": 39.906,
                "latency": 925.46,
                "stderr": 0.486,
                "cost_per_test": 1.619399,
                "token_totals": {
                    "input_tokens": 256899959,
                    "output_tokens": 22298636,
                    "reasoning_tokens": 16662434,
                    "cache_read_tokens": 75246545,
                    "cache_write_tokens": 181603278
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 39.562,
                "latency": 662.269,
                "stderr": 0.865,
                "cost_per_test": 4.149702,
                "token_totals": {
                    "input_tokens": 423716492,
                    "output_tokens": 17794792,
                    "reasoning_tokens": 13531292,
                    "cache_read_tokens": 217525163,
                    "cache_write_tokens": null
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "fireworks/ember-1": {
                "accuracy": 39.26,
                "latency": 798.612,
                "stderr": 2.164,
                "cost_per_test": 1.95445,
                "token_totals": {
                    "input_tokens": 435525965,
                    "output_tokens": 13497155,
                    "reasoning_tokens": 8350801,
                    "cache_read_tokens": 233160321,
                    "cache_write_tokens": null
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "stepfun/step-5-preview": {
                "accuracy": 38.809,
                "latency": 747.585,
                "stderr": 0.368,
                "cost_per_test": 0.619283,
                "token_totals": {
                    "input_tokens": 464025197,
                    "output_tokens": 20770861,
                    "reasoning_tokens": 0,
                    "cache_read_tokens": 254565035,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 1024000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Stepfun",
                "harness": null
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 38.795,
                "latency": 695.588,
                "stderr": 0.232,
                "cost_per_test": 2.409559,
                "token_totals": {
                    "input_tokens": 421005921,
                    "output_tokens": 20173634,
                    "reasoning_tokens": null,
                    "cache_read_tokens": 230989311,
                    "cache_write_tokens": 189800855
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-4-7": {
                "accuracy": 38.647,
                "latency": 360.437,
                "stderr": 0.617,
                "cost_per_test": 4.032269,
                "token_totals": {
                    "input_tokens": 370221415,
                    "output_tokens": 12293417,
                    "reasoning_tokens": null,
                    "cache_read_tokens": 140274986,
                    "cache_write_tokens": 229852824
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "deepseek/deepseek-v4-pro-0813": {
                "accuracy": 38.507,
                "latency": 974.518,
                "stderr": 0.503,
                "cost_per_test": 0.880465,
                "token_totals": {
                    "input_tokens": 419761736,
                    "output_tokens": 30621918,
                    "reasoning_tokens": 25089786,
                    "cache_read_tokens": 218761173,
                    "cache_write_tokens": null
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "openai/gpt-6-luna": {
                "accuracy": 38.433,
                "latency": 897.881,
                "stderr": 0.513,
                "cost_per_test": 0.124338,
                "token_totals": {
                    "input_tokens": 413604271,
                    "output_tokens": 45731435,
                    "reasoning_tokens": 41794427,
                    "cache_read_tokens": 199389488,
                    "cache_write_tokens": 214114651
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3.8-max": {
                "accuracy": 38.229,
                "latency": 1174.928,
                "stderr": 0.581,
                "cost_per_test": 1.236977,
                "token_totals": {
                    "input_tokens": 313950822,
                    "output_tokens": 28653788,
                    "reasoning_tokens": 23202069,
                    "cache_read_tokens": 139669632,
                    "cache_write_tokens": null
                },
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "zai/glm-5.2": {
                "accuracy": 38.006,
                "latency": 473.158,
                "stderr": 0.786,
                "cost_per_test": 0.713981,
                "token_totals": {
                    "input_tokens": 271639159,
                    "output_tokens": 12183429,
                    "reasoning_tokens": 7820405,
                    "cache_read_tokens": 98781099,
                    "cache_write_tokens": null
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "alibaba/qwen3.8-27b": {
                "accuracy": 36.837,
                "latency": 621.733,
                "stderr": 2.108,
                "cost_per_test": 0.748532,
                "token_totals": {
                    "input_tokens": 422396163,
                    "output_tokens": 41880461,
                    "reasoning_tokens": 34693358,
                    "cache_read_tokens": 0,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "deepseek/deepseek-v4-flash-0731": {
                "accuracy": 36.75,
                "latency": 490.245,
                "stderr": 0.572,
                "cost_per_test": 0.276154,
                "token_totals": {
                    "input_tokens": 406112640,
                    "output_tokens": 27423276,
                    "reasoning_tokens": 22726163,
                    "cache_read_tokens": 213584768,
                    "cache_write_tokens": null
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "grok/grok-4.5": {
                "accuracy": 36.743,
                "latency": 289.014,
                "stderr": 0.382,
                "cost_per_test": 1.156495,
                "token_totals": {
                    "input_tokens": 346429154,
                    "output_tokens": 13008523,
                    "reasoning_tokens": 7208719,
                    "cache_read_tokens": 193601579,
                    "cache_write_tokens": null
                },
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "minimax/MiniMax-M3": {
                "accuracy": 36.693,
                "latency": 496.894,
                "stderr": 0.157,
                "cost_per_test": 0.32345,
                "token_totals": {
                    "input_tokens": 302569216,
                    "output_tokens": 10069655,
                    "reasoning_tokens": null,
                    "cache_read_tokens": 126041678,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 512000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "openai/gpt-6-sol": {
                "accuracy": 35.981,
                "latency": 539.907,
                "stderr": 0.434,
                "cost_per_test": 2.122121,
                "token_totals": {
                    "input_tokens": 416329027,
                    "output_tokens": 22746494,
                    "reasoning_tokens": 18774690,
                    "cache_read_tokens": 155710451,
                    "cache_write_tokens": 260528040
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3.5-flash-lite": {
                "accuracy": 35.265,
                "latency": 142.913,
                "stderr": 0.379,
                "cost_per_test": 0.389575,
                "token_totals": {
                    "input_tokens": 670526775,
                    "output_tokens": 10900400,
                    "reasoning_tokens": 7520808,
                    "cache_read_tokens": 196667284,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3.7-max": {
                "accuracy": 34.633,
                "latency": 635.178,
                "stderr": 0.859,
                "cost_per_test": 0.553214,
                "token_totals": {
                    "input_tokens": 104973358,
                    "output_tokens": 3989677,
                    "reasoning_tokens": 2814264,
                    "cache_read_tokens": 19536384,
                    "cache_write_tokens": null
                },
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "thinkingmachines/inkling": {
                "accuracy": 33.428,
                "latency": 1082.125,
                "stderr": 1.024,
                "cost_per_test": 0.996264,
                "token_totals": {
                    "input_tokens": 882963514,
                    "output_tokens": 24911364,
                    "reasoning_tokens": null,
                    "cache_read_tokens": 647224491,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "zai/glm-5.1": {
                "accuracy": 32.68,
                "latency": 924.078,
                "stderr": 0.707,
                "cost_per_test": 0.618185,
                "token_totals": {
                    "input_tokens": 498811983,
                    "output_tokens": 15910910,
                    "reasoning_tokens": 12077030,
                    "cache_read_tokens": 339429333,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 32.584,
                "latency": 1629.639,
                "stderr": 0.465,
                "cost_per_test": 1.141587,
                "token_totals": {
                    "input_tokens": 846905588,
                    "output_tokens": 23540852,
                    "reasoning_tokens": 0,
                    "cache_read_tokens": 488314911,
                    "cache_write_tokens": null
                },
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 32.369,
                "latency": 2153.503,
                "stderr": 0.155,
                "cost_per_test": 1.201815,
                "token_totals": {
                    "input_tokens": 609985753,
                    "output_tokens": 72941360,
                    "reasoning_tokens": 70078573,
                    "cache_read_tokens": 362827776,
                    "cache_write_tokens": null
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 31.448,
                "latency": 1526.183,
                "stderr": 0.715,
                "cost_per_test": 0.879086,
                "token_totals": {
                    "input_tokens": 508078472,
                    "output_tokens": 22058372,
                    "reasoning_tokens": 17233163,
                    "cache_read_tokens": 284032811,
                    "cache_write_tokens": null
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 30.879,
                "latency": 208.258,
                "stderr": 1.348,
                "cost_per_test": 1.591906,
                "token_totals": {
                    "input_tokens": 365193664,
                    "output_tokens": 7730374,
                    "reasoning_tokens": 5787767,
                    "cache_read_tokens": 123797501,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 30.204,
                "latency": 246.543,
                "stderr": 0.675,
                "cost_per_test": 0.570657,
                "token_totals": {
                    "input_tokens": 657633680,
                    "output_tokens": 15218094,
                    "reasoning_tokens": 12099810,
                    "cache_read_tokens": 261501224,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 29.35,
                "latency": 611.907,
                "stderr": 0.67,
                "cost_per_test": 0.814379,
                "token_totals": {
                    "input_tokens": 488661479,
                    "output_tokens": 11569301,
                    "reasoning_tokens": 8412225,
                    "cache_read_tokens": 0,
                    "cache_write_tokens": null
                },
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "thinkingmachines/inkling-small": {
                "accuracy": 29.151,
                "latency": 994.255,
                "stderr": 0.342,
                "cost_per_test": 0.230271,
                "token_totals": {
                    "input_tokens": 696484546,
                    "output_tokens": 28023249,
                    "reasoning_tokens": null,
                    "cache_read_tokens": 580562347,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 262000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "xiaomi/mimo-v2.5-pro": {
                "accuracy": 28.848,
                "latency": 497.373,
                "stderr": 1.247,
                "cost_per_test": 0.206445,
                "token_totals": {
                    "input_tokens": 507567154,
                    "output_tokens": 8732144,
                    "reasoning_tokens": 5054819,
                    "cache_read_tokens": 314066453,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "grok/grok-4.3": {
                "accuracy": 26.373,
                "latency": 615.499,
                "stderr": 0.703,
                "cost_per_test": 0.848345,
                "token_totals": {
                    "input_tokens": 348106276,
                    "output_tokens": 15533931,
                    "reasoning_tokens": 12549303,
                    "cache_read_tokens": 125272960,
                    "cache_write_tokens": null
                },
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 26.022,
                "latency": 334.538,
                "stderr": 0.989,
                "cost_per_test": 0.15925,
                "token_totals": {
                    "input_tokens": 584481947,
                    "output_tokens": 10638763,
                    "reasoning_tokens": 7882802,
                    "cache_read_tokens": 325180587,
                    "cache_write_tokens": null
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3.7-plus": {
                "accuracy": 25.816,
                "latency": 542.045,
                "stderr": 1.318,
                "cost_per_test": 0.357661,
                "token_totals": {
                    "input_tokens": 446705484,
                    "output_tokens": 9996981,
                    "reasoning_tokens": 7272392,
                    "cache_read_tokens": 192899072,
                    "cache_write_tokens": null
                },
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "nvidia/nemotron-3-ultra-550b-a55b": {
                "accuracy": 24.802,
                "latency": 253.882,
                "stderr": 0.728,
                "cost_per_test": null,
                "token_totals": {
                    "input_tokens": 408960062,
                    "output_tokens": 7705379,
                    "reasoning_tokens": 0,
                    "cache_read_tokens": 0,
                    "cache_write_tokens": null
                },
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Nvidia",
                "harness": null
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 23.799,
                "latency": 751.754,
                "stderr": 1.923,
                "cost_per_test": 0.294389,
                "token_totals": {
                    "input_tokens": 324471186,
                    "output_tokens": 7830906,
                    "reasoning_tokens": 5301673,
                    "cache_read_tokens": 171400704,
                    "cache_write_tokens": null
                },
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "xiaomi/mimo-v2.5": {
                "accuracy": 23.785,
                "latency": 339.26,
                "stderr": 0.783,
                "cost_per_test": 0.08693,
                "token_totals": {
                    "input_tokens": 616264480,
                    "output_tokens": 8440426,
                    "reasoning_tokens": 5101469,
                    "cache_read_tokens": 360948032,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "mistralai/mistral-medium-3.5": {
                "accuracy": 21.632,
                "latency": 768.405,
                "stderr": 0.963,
                "cost_per_test": 1.908513,
                "token_totals": {
                    "input_tokens": 448123641,
                    "output_tokens": 24450234,
                    "reasoning_tokens": null,
                    "cache_read_tokens": null,
                    "cache_write_tokens": null
                },
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 19.83,
                "latency": 176.769,
                "stderr": 1.293,
                "cost_per_test": 0.602129,
                "token_totals": {
                    "input_tokens": 314188143,
                    "output_tokens": 7984201,
                    "reasoning_tokens": null,
                    "cache_read_tokens": 140237697,
                    "cache_write_tokens": 172251987
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "ant/ling-3.0-flash-2607": {
                "accuracy": 19.571,
                "latency": 157.02,
                "stderr": 0.115,
                "cost_per_test": 0.05047,
                "token_totals": {
                    "input_tokens": 593259509,
                    "output_tokens": 13185323,
                    "reasoning_tokens": null,
                    "cache_read_tokens": 426536000,
                    "cache_write_tokens": null
                },
                "temperature": 0.6,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 19.303,
                "latency": 54.778,
                "stderr": 0.526,
                "cost_per_test": 0.142482,
                "token_totals": {
                    "input_tokens": 309249776,
                    "output_tokens": 4518540,
                    "reasoning_tokens": 3033245,
                    "cache_read_tokens": 88769598,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 17.492,
                "latency": 351.413,
                "stderr": 0.099,
                "cost_per_test": 0.791809,
                "token_totals": {
                    "input_tokens": 181695255,
                    "output_tokens": 12341665,
                    "reasoning_tokens": 9833132,
                    "cache_read_tokens": 66807493,
                    "cache_write_tokens": null
                },
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 17.135,
                "latency": 803.528,
                "stderr": 0.187,
                "cost_per_test": 0.18348,
                "token_totals": {
                    "input_tokens": 573686662,
                    "output_tokens": 15857919,
                    "reasoning_tokens": null,
                    "cache_read_tokens": 404654261,
                    "cache_write_tokens": 9164196
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "poolside/laguna-m.1": {
                "accuracy": 14.419,
                "latency": 1223.035,
                "stderr": 0.866,
                "cost_per_test": 0.818192,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "fireworks/nemotron-lightning-3p5-30b-a3b": {
                "accuracy": 10.289,
                "latency": 539.61,
                "stderr": 0.493,
                "cost_per_test": 0.028554,
                "token_totals": {
                    "input_tokens": 335136335,
                    "output_tokens": 19277054,
                    "reasoning_tokens": null,
                    "cache_read_tokens": null,
                    "cache_write_tokens": null
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "inception/mercury-2.5": {
                "accuracy": 9.369,
                "latency": 92.482,
                "stderr": 0.269,
                "cost_per_test": 0.106441,
                "token_totals": {
                    "input_tokens": 376083870,
                    "output_tokens": 5831330,
                    "reasoning_tokens": 4955938,
                    "cache_read_tokens": 191920653,
                    "cache_write_tokens": null
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Inception",
                "harness": null
            },
            "poolside/laguna-xs.2": {
                "accuracy": 6.794,
                "latency": 526.624,
                "stderr": 0.353,
                "cost_per_test": 0.19216,
                "token_totals": {
                    "input_tokens": 1269502184,
                    "output_tokens": 8838931,
                    "reasoning_tokens": 5301032,
                    "cache_read_tokens": 283700016,
                    "cache_write_tokens": 0
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "cohere/command-a-plus-05-2026": {
                "accuracy": 2.761,
                "latency": 566.745,
                "stderr": 0.539,
                "cost_per_test": 1.63003,
                "token_totals": {
                    "input_tokens": 212326507,
                    "output_tokens": 20269701,
                    "reasoning_tokens": 19053167,
                    "cache_read_tokens": 0,
                    "cache_write_tokens": null
                },
                "temperature": 0.9,
                "top_p": 0.95,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            }
        },
        "general_qualitative_analysis": {
            "meta/muse_spark_1_3_max": {
                "accuracy": 83.373,
                "latency": 199.524,
                "stderr": 4.878,
                "cost_per_test": 0.796294,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "grok/grok-4.6": {
                "accuracy": 77.762,
                "latency": 1367.816,
                "stderr": 0.582,
                "cost_per_test": 1.448195,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "meta/muse_spark_1_2": {
                "accuracy": 77.593,
                "latency": 322.846,
                "stderr": 1.157,
                "cost_per_test": 0.786782,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "meta/muse_spark_1_3": {
                "accuracy": 77.552,
                "latency": 334.646,
                "stderr": 1.72,
                "cost_per_test": 0.767522,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "google/gemini-4-argon": {
                "accuracy": 76.963,
                "latency": 1013.872,
                "stderr": 0.137,
                "cost_per_test": 4.226516,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-fable-5-1": {
                "accuracy": 76.962,
                "latency": 3231.502,
                "stderr": 5.342,
                "cost_per_test": 12.050237,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "zai/glm-5.3-flash": {
                "accuracy": 76.247,
                "latency": 1414.758,
                "stderr": 5.301,
                "cost_per_test": 0.062892,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "anthropic/claude-opus-5": {
                "accuracy": 75.442,
                "latency": 1029.573,
                "stderr": 1.221,
                "cost_per_test": 7.880191,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "xiaomi/mimo-v2.6-pro": {
                "accuracy": 75.128,
                "latency": 512.273,
                "stderr": 0.799,
                "cost_per_test": 0.236136,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "xiaomi/mimo-v2.6-flash": {
                "accuracy": 72.532,
                "latency": 434.118,
                "stderr": 1.714,
                "cost_per_test": 0.087241,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 71.557,
                "latency": 705.169,
                "stderr": 1.493,
                "cost_per_test": 2.368893,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-4.5": {
                "accuracy": 71.132,
                "latency": 245.415,
                "stderr": 2.382,
                "cost_per_test": 1.166846,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-fable-5": {
                "accuracy": 70.324,
                "latency": 1140.824,
                "stderr": 0.379,
                "cost_per_test": 11.155704,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "zai/glm-5.3": {
                "accuracy": 70.172,
                "latency": 1204.616,
                "stderr": 5.827,
                "cost_per_test": 1.336081,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "tencent/hy4-preview": {
                "accuracy": 69.393,
                "latency": 1795.712,
                "stderr": 1.406,
                "cost_per_test": 0.891678,
                "temperature": 0.9,
                "top_p": 1.0,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Tencent",
                "harness": null
            },
            "grok/grok-4.7": {
                "accuracy": 68.736,
                "latency": 786.67,
                "stderr": 0.745,
                "cost_per_test": 2.405329,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "zai/glm-5.2": {
                "accuracy": 68.73,
                "latency": 390.229,
                "stderr": 3.079,
                "cost_per_test": 0.64217,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "openai/gpt-6-astra": {
                "accuracy": 68.425,
                "latency": 1070.103,
                "stderr": 5.687,
                "cost_per_test": 9.77377,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3.8-flash": {
                "accuracy": 68.07,
                "latency": 549.434,
                "stderr": 0.385,
                "cost_per_test": 1.837672,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.6-terra": {
                "accuracy": 67.802,
                "latency": 1944.23,
                "stderr": 5.94,
                "cost_per_test": 4.427551,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-5-5": {
                "accuracy": 67.773,
                "latency": 3659.306,
                "stderr": 1.711,
                "cost_per_test": 15.192056,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "deepseek/deepseek-v4.1-flash": {
                "accuracy": 67.397,
                "latency": 400.534,
                "stderr": 2.372,
                "cost_per_test": 0.228616,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "deepseek/deepseek-v4-pro-0813": {
                "accuracy": 67.367,
                "latency": 862.657,
                "stderr": 0.446,
                "cost_per_test": 0.915589,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "ant/ling-3.0-flash-af-rc3": {
                "accuracy": 67.154,
                "latency": 495.562,
                "stderr": 0.332,
                "cost_per_test": 0.056342,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "anthropic/claude-sonnet-5-5": {
                "accuracy": 67.06,
                "latency": 3366.451,
                "stderr": 0.137,
                "cost_per_test": 10.63213,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.6-luna": {
                "accuracy": 66.558,
                "latency": 773.715,
                "stderr": 1.08,
                "cost_per_test": 0.272581,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-4-8": {
                "accuracy": 66.414,
                "latency": 842.804,
                "stderr": 2.134,
                "cost_per_test": 5.273958,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "kimi/kimi-k3": {
                "accuracy": 66.408,
                "latency": 634.089,
                "stderr": 0.803,
                "cost_per_test": 1.842312,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "meta/muse_spark_1_1": {
                "accuracy": 66.365,
                "latency": 409.17,
                "stderr": 1.941,
                "cost_per_test": 0.746224,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 65.559,
                "latency": 717.08,
                "stderr": 1.39,
                "cost_per_test": 4.530392,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-4-7": {
                "accuracy": 65.432,
                "latency": 351.773,
                "stderr": 0.781,
                "cost_per_test": 4.402011,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "mistralai/mistral-large-4": {
                "accuracy": 65.399,
                "latency": 748.301,
                "stderr": 2.275,
                "cost_per_test": 1.103219,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "google/gemini-3.7-flash": {
                "accuracy": 65.289,
                "latency": 457.509,
                "stderr": 2.24,
                "cost_per_test": 1.310653,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.6-sol": {
                "accuracy": 64.933,
                "latency": 1605.836,
                "stderr": 3.336,
                "cost_per_test": 1.147273,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-sonnet-5": {
                "accuracy": 64.67,
                "latency": 1344.312,
                "stderr": 1.631,
                "cost_per_test": 3.682439,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "stepfun/step-5-preview": {
                "accuracy": 63.818,
                "latency": 800.231,
                "stderr": 3.451,
                "cost_per_test": 0.691449,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 1024000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Stepfun",
                "harness": null
            },
            "zai/glm-5.1": {
                "accuracy": 63.665,
                "latency": 761.661,
                "stderr": 0.193,
                "cost_per_test": 0.494021,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "alibaba/qwen3.7-max": {
                "accuracy": 63.521,
                "latency": 653.474,
                "stderr": 1.393,
                "cost_per_test": 1.909371,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "alibaba/qwen3.8-max": {
                "accuracy": 63.375,
                "latency": 1443.121,
                "stderr": 1.97,
                "cost_per_test": 1.531101,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/gpt-6.1-sol": {
                "accuracy": 63.205,
                "latency": 1375.679,
                "stderr": 0.97,
                "cost_per_test": 2.238549,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "thinkingmachines/inkling": {
                "accuracy": 62.915,
                "latency": 1077.969,
                "stderr": 5.063,
                "cost_per_test": 1.000908,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "alibaba/qwen3.8-27b": {
                "accuracy": 62.107,
                "latency": 858.513,
                "stderr": 6.542,
                "cost_per_test": 0.696619,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "anthropic/claude-haiku-5-5": {
                "accuracy": 61.879,
                "latency": 3409.117,
                "stderr": 6.22,
                "cost_per_test": 3.048859,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "fireworks/ember-1": {
                "accuracy": 61.744,
                "latency": 1571.631,
                "stderr": 6.052,
                "cost_per_test": 2.433495,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "minimax/MiniMax-M3": {
                "accuracy": 61.691,
                "latency": 498.741,
                "stderr": 2.969,
                "cost_per_test": 0.352382,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 512000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "google/gemini-3.6-flash": {
                "accuracy": 60.456,
                "latency": 501.008,
                "stderr": 1.077,
                "cost_per_test": 1.316198,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "deepseek/deepseek-v4-flash-0731": {
                "accuracy": 60.449,
                "latency": 584.887,
                "stderr": 0.925,
                "cost_per_test": 0.337187,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 60.341,
                "latency": 1682.108,
                "stderr": 1.505,
                "cost_per_test": 0.959616,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "nvidia/nemotron-3-ultra-550b-a55b": {
                "accuracy": 58.487,
                "latency": 231.138,
                "stderr": 3.971,
                "cost_per_test": null,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Nvidia",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 57.295,
                "latency": 1520.92,
                "stderr": 2.228,
                "cost_per_test": 1.003368,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "google/gemini-3.5-flash": {
                "accuracy": 56.832,
                "latency": 287.438,
                "stderr": 0.418,
                "cost_per_test": 2.622948,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "thinkingmachines/inkling-small": {
                "accuracy": 56.345,
                "latency": 1562.68,
                "stderr": 1.581,
                "cost_per_test": 0.162864,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 262000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "openai/gpt-6-luna": {
                "accuracy": 54.297,
                "latency": 934.658,
                "stderr": 3.156,
                "cost_per_test": 0.128458,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-6-sol": {
                "accuracy": 53.615,
                "latency": 782.017,
                "stderr": 1.824,
                "cost_per_test": 2.899374,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 53.222,
                "latency": 974.308,
                "stderr": 6.175,
                "cost_per_test": 0.285177,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "grok/grok-4.3": {
                "accuracy": 53.198,
                "latency": 749.693,
                "stderr": 0.902,
                "cost_per_test": 0.78556,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 51.94,
                "latency": 608.006,
                "stderr": 1.346,
                "cost_per_test": 1.906543,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "google/gemini-3.5-flash-lite": {
                "accuracy": 51.11,
                "latency": 150.92,
                "stderr": 0.239,
                "cost_per_test": 0.275972,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "xiaomi/mimo-v2.5-pro": {
                "accuracy": 50.768,
                "latency": 426.85,
                "stderr": 4.219,
                "cost_per_test": 0.210443,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "xiaomi/mimo-v2.5": {
                "accuracy": 47.992,
                "latency": 278.393,
                "stderr": 2.089,
                "cost_per_test": 0.097278,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "mistralai/mistral-medium-3.5": {
                "accuracy": 47.935,
                "latency": 618.117,
                "stderr": 2.346,
                "cost_per_test": 1.416491,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 46.366,
                "latency": 2096.642,
                "stderr": 0.966,
                "cost_per_test": 1.273272,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 46.286,
                "latency": 344.238,
                "stderr": 2.19,
                "cost_per_test": 0.160073,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3.7-plus": {
                "accuracy": 45.979,
                "latency": 535.085,
                "stderr": 1.143,
                "cost_per_test": 0.336957,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 45.874,
                "latency": 388.485,
                "stderr": 0.564,
                "cost_per_test": 0.67815,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 45.206,
                "latency": 131.886,
                "stderr": 2.644,
                "cost_per_test": 1.237141,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 45.11,
                "latency": 208.374,
                "stderr": 2.169,
                "cost_per_test": 0.61919,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 44.608,
                "latency": 157.579,
                "stderr": 0.446,
                "cost_per_test": 0.524815,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "ant/ling-3.0-flash-2607": {
                "accuracy": 43.203,
                "latency": 126.825,
                "stderr": 2.793,
                "cost_per_test": 0.041696,
                "temperature": 0.6,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 42.376,
                "latency": 438.133,
                "stderr": 0.992,
                "cost_per_test": 0.102708,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "poolside/laguna-m.1": {
                "accuracy": 36.472,
                "latency": 941.492,
                "stderr": 3.119,
                "cost_per_test": 0.382036,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 36.184,
                "latency": 42.875,
                "stderr": 0.795,
                "cost_per_test": 0.128348,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "fireworks/nemotron-lightning-3p5-30b-a3b": {
                "accuracy": 33.048,
                "latency": 1654.235,
                "stderr": 0.485,
                "cost_per_test": 0.026094,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "inception/mercury-2.5": {
                "accuracy": 29.476,
                "latency": 94.169,
                "stderr": 1.076,
                "cost_per_test": 0.059496,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Inception",
                "harness": null
            },
            "poolside/laguna-xs.2": {
                "accuracy": 28.879,
                "latency": 306.58,
                "stderr": 1.489,
                "cost_per_test": 0.122583,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "cohere/command-a-plus-05-2026": {
                "accuracy": 22.257,
                "latency": 391.0,
                "stderr": 3.204,
                "cost_per_test": 0.945316,
                "temperature": 0.9,
                "top_p": 0.95,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            }
        },
        "precedents": {
            "google/gemini-4-argon": {
                "accuracy": 49.834,
                "latency": 1674.662,
                "stderr": 0.628,
                "cost_per_test": 7.237106,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.5-flash": {
                "accuracy": 36.365,
                "latency": 447.419,
                "stderr": 2.688,
                "cost_per_test": 3.899099,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.8-flash": {
                "accuracy": 36.141,
                "latency": 378.489,
                "stderr": 2.185,
                "cost_per_test": 3.153918,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "meta/muse_spark_1_2": {
                "accuracy": 35.362,
                "latency": 466.764,
                "stderr": 0.68,
                "cost_per_test": 1.265096,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "meta/muse_spark_1_3_max": {
                "accuracy": 35.223,
                "latency": 323.4,
                "stderr": 6.774,
                "cost_per_test": 1.121068,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "anthropic/claude-opus-5-5": {
                "accuracy": 34.59,
                "latency": 2586.845,
                "stderr": 0.538,
                "cost_per_test": 11.856107,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.7-flash": {
                "accuracy": 33.813,
                "latency": 185.067,
                "stderr": 2.431,
                "cost_per_test": 2.090224,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "meta/muse_spark_1_1": {
                "accuracy": 33.81,
                "latency": 674.897,
                "stderr": 2.472,
                "cost_per_test": 1.036058,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "meta/muse_spark_1_3": {
                "accuracy": 33.095,
                "latency": 460.491,
                "stderr": 1.902,
                "cost_per_test": 1.092387,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "anthropic/claude-fable-5-1": {
                "accuracy": 32.202,
                "latency": 988.478,
                "stderr": 6.574,
                "cost_per_test": 10.708652,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-sonnet-5-5": {
                "accuracy": 31.697,
                "latency": 2565.821,
                "stderr": 2.195,
                "cost_per_test": 9.316281,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "xiaomi/mimo-v2.6-pro": {
                "accuracy": 30.879,
                "latency": 924.628,
                "stderr": 2.147,
                "cost_per_test": 0.2749,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "google/gemini-3.6-flash": {
                "accuracy": 30.76,
                "latency": 577.934,
                "stderr": 1.034,
                "cost_per_test": 1.833435,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "tencent/hy4-preview": {
                "accuracy": 30.601,
                "latency": 1840.379,
                "stderr": 2.714,
                "cost_per_test": 0.738743,
                "temperature": 0.9,
                "top_p": 1.0,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Tencent",
                "harness": null
            },
            "openai/gpt-6-astra": {
                "accuracy": 30.289,
                "latency": 1034.651,
                "stderr": 6.403,
                "cost_per_test": 11.362177,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "xiaomi/mimo-v2.6-flash": {
                "accuracy": 30.251,
                "latency": 521.5,
                "stderr": 2.328,
                "cost_per_test": 0.092173,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "fireworks/ember-1": {
                "accuracy": 29.193,
                "latency": 1145.834,
                "stderr": 6.476,
                "cost_per_test": 3.728924,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "openai/gpt-5.6-luna": {
                "accuracy": 29.12,
                "latency": 1409.266,
                "stderr": 0.072,
                "cost_per_test": 0.529871,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "ant/ling-3.0-flash-af-rc3": {
                "accuracy": 28.971,
                "latency": 473.129,
                "stderr": 1.627,
                "cost_per_test": 0.058545,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "anthropic/claude-sonnet-5": {
                "accuracy": 28.774,
                "latency": 1101.385,
                "stderr": 1.863,
                "cost_per_test": 4.375718,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "zai/glm-5.3-flash": {
                "accuracy": 28.582,
                "latency": 913.812,
                "stderr": 6.337,
                "cost_per_test": 0.053872,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "openai/gpt-6-sol": {
                "accuracy": 28.499,
                "latency": 839.987,
                "stderr": 1.713,
                "cost_per_test": 3.902195,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "deepseek/deepseek-v4.1-flash": {
                "accuracy": 28.219,
                "latency": 563.78,
                "stderr": 3.6,
                "cost_per_test": 0.306545,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "anthropic/claude-fable-5": {
                "accuracy": 28.097,
                "latency": 890.071,
                "stderr": 2.764,
                "cost_per_test": 9.631473,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-6-luna": {
                "accuracy": 27.379,
                "latency": 1470.22,
                "stderr": 1.645,
                "cost_per_test": 0.214229,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-haiku-5-5": {
                "accuracy": 26.976,
                "latency": 2427.307,
                "stderr": 6.289,
                "cost_per_test": 2.195932,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.6-sol": {
                "accuracy": 26.192,
                "latency": 1935.492,
                "stderr": 0.847,
                "cost_per_test": 2.11729,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "mistralai/mistral-large-4": {
                "accuracy": 26.18,
                "latency": 2392.646,
                "stderr": 2.202,
                "cost_per_test": 1.982866,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "google/gemini-3.5-flash-lite": {
                "accuracy": 25.972,
                "latency": 540.069,
                "stderr": 0.772,
                "cost_per_test": 0.58475,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-4.7": {
                "accuracy": 25.944,
                "latency": 1594.782,
                "stderr": 1.609,
                "cost_per_test": 5.674192,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-opus-5": {
                "accuracy": 25.87,
                "latency": 768.873,
                "stderr": 0.82,
                "cost_per_test": 6.731111,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-6.1-sol": {
                "accuracy": 25.785,
                "latency": 1361.448,
                "stderr": 1.548,
                "cost_per_test": 2.645364,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-4-8": {
                "accuracy": 25.627,
                "latency": 1099.702,
                "stderr": 3.199,
                "cost_per_test": 6.192772,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "zai/glm-5.3": {
                "accuracy": 25.127,
                "latency": 1733.893,
                "stderr": 6.174,
                "cost_per_test": 1.384529,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "openai/gpt-5.6-terra": {
                "accuracy": 24.845,
                "latency": 2481.622,
                "stderr": 6.123,
                "cost_per_test": 6.593612,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "thinkingmachines/inkling": {
                "accuracy": 24.758,
                "latency": 2032.818,
                "stderr": 3.663,
                "cost_per_test": 1.32866,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "deepseek/deepseek-v4-flash-0731": {
                "accuracy": 24.08,
                "latency": 699.303,
                "stderr": 2.16,
                "cost_per_test": 0.41556,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "kimi/kimi-k3": {
                "accuracy": 23.634,
                "latency": 449.804,
                "stderr": 2.42,
                "cost_per_test": 3.338269,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "minimax/MiniMax-M3": {
                "accuracy": 23.24,
                "latency": 804.141,
                "stderr": 1.574,
                "cost_per_test": 0.571079,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 512000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 22.111,
                "latency": 3206.023,
                "stderr": 2.142,
                "cost_per_test": 1.890767,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4.6": {
                "accuracy": 21.918,
                "latency": 1612.355,
                "stderr": 1.579,
                "cost_per_test": 3.017002,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 21.894,
                "latency": 1110.441,
                "stderr": 1.895,
                "cost_per_test": 7.463278,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3.8-27b": {
                "accuracy": 21.603,
                "latency": 782.899,
                "stderr": 5.626,
                "cost_per_test": 1.002792,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "anthropic/claude-opus-4-7": {
                "accuracy": 21.497,
                "latency": 570.859,
                "stderr": 1.086,
                "cost_per_test": 7.128385,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "stepfun/step-5-preview": {
                "accuracy": 21.142,
                "latency": 1404.168,
                "stderr": 0.79,
                "cost_per_test": 1.077508,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 1024000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Stepfun",
                "harness": null
            },
            "deepseek/deepseek-v4-pro-0813": {
                "accuracy": 20.89,
                "latency": 1435.1,
                "stderr": 2.387,
                "cost_per_test": 1.22663,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 20.83,
                "latency": 532.115,
                "stderr": 1.51,
                "cost_per_test": 0.276849,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 20.816,
                "latency": 344.984,
                "stderr": 3.014,
                "cost_per_test": 3.095248,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "xiaomi/mimo-v2.5-pro": {
                "accuracy": 20.558,
                "latency": 704.058,
                "stderr": 2.335,
                "cost_per_test": 0.298475,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "thinkingmachines/inkling-small": {
                "accuracy": 20.227,
                "latency": 1071.691,
                "stderr": 1.354,
                "cost_per_test": 0.30273,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 262000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "alibaba/qwen3.7-max": {
                "accuracy": 20.04,
                "latency": 1024.009,
                "stderr": 2.03,
                "cost_per_test": 3.170073,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "alibaba/qwen3.8-max": {
                "accuracy": 19.494,
                "latency": 2018.307,
                "stderr": 3.805,
                "cost_per_test": 1.762214,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 19.459,
                "latency": 2890.369,
                "stderr": 1.755,
                "cost_per_test": 1.843727,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "zai/glm-5.2": {
                "accuracy": 19.379,
                "latency": 724.685,
                "stderr": 1.262,
                "cost_per_test": 1.122176,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "grok/grok-4.5": {
                "accuracy": 19.372,
                "latency": 496.806,
                "stderr": 0.428,
                "cost_per_test": 2.127011,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 18.923,
                "latency": 1243.255,
                "stderr": 1.821,
                "cost_per_test": 4.197491,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 18.138,
                "latency": 453.499,
                "stderr": 0.789,
                "cost_per_test": 0.955464,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3.7-plus": {
                "accuracy": 18.067,
                "latency": 1002.639,
                "stderr": 2.655,
                "cost_per_test": 0.525511,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "zai/glm-5.1": {
                "accuracy": 17.764,
                "latency": 1623.858,
                "stderr": 2.425,
                "cost_per_test": 0.997596,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "xiaomi/mimo-v2.5": {
                "accuracy": 17.206,
                "latency": 413.518,
                "stderr": 1.229,
                "cost_per_test": 0.114392,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 17.137,
                "latency": 821.432,
                "stderr": 1.22,
                "cost_per_test": 1.054446,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 15.6,
                "latency": 2630.599,
                "stderr": 0.961,
                "cost_per_test": 1.31745,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 15.536,
                "latency": 448.504,
                "stderr": 1.652,
                "cost_per_test": 1.29945,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "grok/grok-4.3": {
                "accuracy": 15.265,
                "latency": 709.986,
                "stderr": 1.76,
                "cost_per_test": 1.289733,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "ant/ling-3.0-flash-2607": {
                "accuracy": 14.689,
                "latency": 169.818,
                "stderr": 1.676,
                "cost_per_test": 0.056116,
                "temperature": 0.6,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 14.594,
                "latency": 1099.014,
                "stderr": 5.038,
                "cost_per_test": 0.435205,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "mistralai/mistral-medium-3.5": {
                "accuracy": 12.901,
                "latency": 1118.384,
                "stderr": 2.211,
                "cost_per_test": 2.287472,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "nvidia/nemotron-3-ultra-550b-a55b": {
                "accuracy": 12.816,
                "latency": 333.089,
                "stderr": 3.421,
                "cost_per_test": null,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Nvidia",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 11.943,
                "latency": 1168.843,
                "stderr": 0.943,
                "cost_per_test": 0.209794,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 11.808,
                "latency": 60.876,
                "stderr": 0.741,
                "cost_per_test": 0.143796,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "poolside/laguna-m.1": {
                "accuracy": 11.105,
                "latency": 1395.378,
                "stderr": 1.271,
                "cost_per_test": 0.824051,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 11.012,
                "latency": 229.949,
                "stderr": 2.202,
                "cost_per_test": 0.667183,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "inception/mercury-2.5": {
                "accuracy": 7.448,
                "latency": 131.26,
                "stderr": 0.684,
                "cost_per_test": 0.097081,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Inception",
                "harness": null
            },
            "poolside/laguna-xs.2": {
                "accuracy": 5.79,
                "latency": 540.253,
                "stderr": 0.903,
                "cost_per_test": 0.205593,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "fireworks/nemotron-lightning-3p5-30b-a3b": {
                "accuracy": 4.468,
                "latency": 753.804,
                "stderr": 1.178,
                "cost_per_test": 0.023679,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "cohere/command-a-plus-05-2026": {
                "accuracy": 2.395,
                "latency": 585.071,
                "stderr": 0.666,
                "cost_per_test": 1.469353,
                "temperature": 0.9,
                "top_p": 0.95,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            }
        },
        "general_quantitative_analysis": {
            "google/gemini-3.7-flash": {
                "accuracy": 81.81,
                "latency": 599.427,
                "stderr": 0.868,
                "cost_per_test": 1.564066,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.6-flash": {
                "accuracy": 81.808,
                "latency": 642.213,
                "stderr": 0.821,
                "cost_per_test": 1.573127,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-4-argon": {
                "accuracy": 81.515,
                "latency": 834.585,
                "stderr": 1.183,
                "cost_per_test": 4.027599,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.8-flash": {
                "accuracy": 80.851,
                "latency": 522.69,
                "stderr": 0.595,
                "cost_per_test": 1.978613,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-fable-5": {
                "accuracy": 80.751,
                "latency": 709.258,
                "stderr": 1.207,
                "cost_per_test": 8.098129,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "xiaomi/mimo-v2.6-pro": {
                "accuracy": 79.823,
                "latency": 488.736,
                "stderr": 0.499,
                "cost_per_test": 0.220661,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "anthropic/claude-fable-5-1": {
                "accuracy": 79.746,
                "latency": 2852.626,
                "stderr": 4.886,
                "cost_per_test": 8.493132,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-sonnet-5-5": {
                "accuracy": 79.44,
                "latency": 1525.614,
                "stderr": 1.343,
                "cost_per_test": 5.661267,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-5-5": {
                "accuracy": 79.293,
                "latency": 1506.107,
                "stderr": 0.469,
                "cost_per_test": 8.067548,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-6-astra": {
                "accuracy": 78.779,
                "latency": 473.013,
                "stderr": 4.926,
                "cost_per_test": 5.477409,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-5": {
                "accuracy": 78.502,
                "latency": 474.835,
                "stderr": 0.938,
                "cost_per_test": 4.848634,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "meta/muse_spark_1_2": {
                "accuracy": 78.397,
                "latency": 341.327,
                "stderr": 0.769,
                "cost_per_test": 0.749545,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "meta/muse_spark_1_3_max": {
                "accuracy": 77.994,
                "latency": 164.397,
                "stderr": 5.099,
                "cost_per_test": 0.689332,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "google/gemini-3.5-flash": {
                "accuracy": 77.597,
                "latency": 304.655,
                "stderr": 1.028,
                "cost_per_test": 2.413434,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "ant/ling-3.0-flash-af-rc3": {
                "accuracy": 77.243,
                "latency": 343.909,
                "stderr": 0.814,
                "cost_per_test": 0.044226,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "meta/muse_spark_1_1": {
                "accuracy": 76.845,
                "latency": 418.705,
                "stderr": 0.97,
                "cost_per_test": 0.78633,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "meta/muse_spark_1_3": {
                "accuracy": 76.764,
                "latency": 363.991,
                "stderr": 1.602,
                "cost_per_test": 0.733456,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "zai/glm-5.3-flash": {
                "accuracy": 76.433,
                "latency": 874.224,
                "stderr": 5.225,
                "cost_per_test": 0.045357,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "mistralai/mistral-large-4": {
                "accuracy": 75.682,
                "latency": 887.323,
                "stderr": 1.24,
                "cost_per_test": 1.086558,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "openai/gpt-5.6-sol": {
                "accuracy": 75.613,
                "latency": 851.845,
                "stderr": 1.885,
                "cost_per_test": 0.724133,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "tencent/hy4-preview": {
                "accuracy": 75.509,
                "latency": 1223.459,
                "stderr": 1.647,
                "cost_per_test": 0.571386,
                "temperature": 0.9,
                "top_p": 1.0,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Tencent",
                "harness": null
            },
            "openai/gpt-5.6-luna": {
                "accuracy": 75.207,
                "latency": 643.221,
                "stderr": 1.135,
                "cost_per_test": 0.234078,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "fireworks/ember-1": {
                "accuracy": 75.087,
                "latency": 1032.167,
                "stderr": 5.289,
                "cost_per_test": 1.51608,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "kimi/kimi-k3": {
                "accuracy": 74.807,
                "latency": 204.787,
                "stderr": 0.853,
                "cost_per_test": 1.675249,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "openai/gpt-6.1-sol": {
                "accuracy": 74.527,
                "latency": 633.576,
                "stderr": 1.331,
                "cost_per_test": 1.352629,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-haiku-5-5": {
                "accuracy": 74.362,
                "latency": 1399.637,
                "stderr": 5.385,
                "cost_per_test": 1.377648,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-sonnet-5": {
                "accuracy": 74.268,
                "latency": 847.356,
                "stderr": 1.12,
                "cost_per_test": 2.726381,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 73.714,
                "latency": 485.345,
                "stderr": 1.754,
                "cost_per_test": 3.745478,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4.6": {
                "accuracy": 73.245,
                "latency": 747.719,
                "stderr": 2.656,
                "cost_per_test": 1.307476,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-5.6-terra": {
                "accuracy": 72.675,
                "latency": 1089.193,
                "stderr": 5.607,
                "cost_per_test": 2.931004,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 72.004,
                "latency": 418.676,
                "stderr": 1.524,
                "cost_per_test": 2.178056,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-4-8": {
                "accuracy": 71.636,
                "latency": 666.365,
                "stderr": 0.975,
                "cost_per_test": 4.002255,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "stepfun/step-5-preview": {
                "accuracy": 71.413,
                "latency": 552.493,
                "stderr": 2.021,
                "cost_per_test": 0.543208,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 1024000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Stepfun",
                "harness": null
            },
            "zai/glm-5.3": {
                "accuracy": 71.079,
                "latency": 858.644,
                "stderr": 5.687,
                "cost_per_test": 1.056972,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "xiaomi/mimo-v2.6-flash": {
                "accuracy": 70.307,
                "latency": 325.275,
                "stderr": 1.22,
                "cost_per_test": 0.073473,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "openai/gpt-6-luna": {
                "accuracy": 70.084,
                "latency": 670.483,
                "stderr": 0.934,
                "cost_per_test": 0.102148,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-4-7": {
                "accuracy": 70.054,
                "latency": 238.532,
                "stderr": 1.051,
                "cost_per_test": 3.744437,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-6-sol": {
                "accuracy": 70.002,
                "latency": 403.982,
                "stderr": 2.026,
                "cost_per_test": 1.779785,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "deepseek/deepseek-v4.1-flash": {
                "accuracy": 69.814,
                "latency": 392.327,
                "stderr": 3.022,
                "cost_per_test": 0.206219,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "grok/grok-4.7": {
                "accuracy": 68.552,
                "latency": 776.754,
                "stderr": 1.79,
                "cost_per_test": 2.281418,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "alibaba/qwen3.8-max": {
                "accuracy": 67.397,
                "latency": 1057.398,
                "stderr": 3.499,
                "cost_per_test": 1.19159,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "alibaba/qwen3.7-max": {
                "accuracy": 67.148,
                "latency": 547.268,
                "stderr": 1.384,
                "cost_per_test": 1.623001,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 67.018,
                "latency": 190.991,
                "stderr": 1.376,
                "cost_per_test": 1.723303,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3.8-27b": {
                "accuracy": 66.938,
                "latency": 548.045,
                "stderr": 5.745,
                "cost_per_test": 0.653698,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "deepseek/deepseek-v4-pro-0813": {
                "accuracy": 66.866,
                "latency": 816.595,
                "stderr": 1.195,
                "cost_per_test": 0.87839,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "minimax/MiniMax-M3": {
                "accuracy": 66.483,
                "latency": 376.513,
                "stderr": 0.823,
                "cost_per_test": 0.308361,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 512000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "google/gemini-3.5-flash-lite": {
                "accuracy": 66.12,
                "latency": 291.898,
                "stderr": 2.586,
                "cost_per_test": 0.427377,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 65.75,
                "latency": 1415.095,
                "stderr": 1.396,
                "cost_per_test": 0.961249,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "zai/glm-5.2": {
                "accuracy": 65.524,
                "latency": 318.272,
                "stderr": 0.325,
                "cost_per_test": 0.625755,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 65.13,
                "latency": 1549.406,
                "stderr": 2.429,
                "cost_per_test": 1.111564,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 64.457,
                "latency": 262.693,
                "stderr": 2.015,
                "cost_per_test": 0.680828,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "thinkingmachines/inkling": {
                "accuracy": 64.229,
                "latency": 1152.174,
                "stderr": 0.945,
                "cost_per_test": 1.109771,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "deepseek/deepseek-v4-flash-0731": {
                "accuracy": 64.032,
                "latency": 441.822,
                "stderr": 0.384,
                "cost_per_test": 0.276167,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 64.025,
                "latency": 1661.07,
                "stderr": 2.039,
                "cost_per_test": 1.07807,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4.5": {
                "accuracy": 63.58,
                "latency": 213.377,
                "stderr": 1.515,
                "cost_per_test": 1.012928,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "zai/glm-5.1": {
                "accuracy": 61.87,
                "latency": 698.813,
                "stderr": 1.774,
                "cost_per_test": 0.556569,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "xiaomi/mimo-v2.5-pro": {
                "accuracy": 61.049,
                "latency": 389.913,
                "stderr": 2.424,
                "cost_per_test": 0.219268,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 58.313,
                "latency": 268.192,
                "stderr": 1.54,
                "cost_per_test": 0.156651,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 58.181,
                "latency": 520.968,
                "stderr": 2.076,
                "cost_per_test": 0.559083,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "thinkingmachines/inkling-small": {
                "accuracy": 56.192,
                "latency": 1453.72,
                "stderr": 3.697,
                "cost_per_test": 0.246056,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 262000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "nvidia/nemotron-3-ultra-550b-a55b": {
                "accuracy": 55.802,
                "latency": 213.531,
                "stderr": 3.002,
                "cost_per_test": null,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Nvidia",
                "harness": null
            },
            "xiaomi/mimo-v2.5": {
                "accuracy": 55.744,
                "latency": 304.449,
                "stderr": 2.949,
                "cost_per_test": 0.098023,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 54.408,
                "latency": 932.143,
                "stderr": 6.054,
                "cost_per_test": 0.341661,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "alibaba/qwen3.7-plus": {
                "accuracy": 52.24,
                "latency": 690.858,
                "stderr": 2.78,
                "cost_per_test": 0.429169,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "grok/grok-4.3": {
                "accuracy": 51.788,
                "latency": 629.819,
                "stderr": 0.465,
                "cost_per_test": 0.871512,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 49.932,
                "latency": 54.088,
                "stderr": 0.284,
                "cost_per_test": 0.151645,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 46.028,
                "latency": 146.919,
                "stderr": 2.926,
                "cost_per_test": 0.69854,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "mistralai/mistral-medium-3.5": {
                "accuracy": 44.752,
                "latency": 761.631,
                "stderr": 2.387,
                "cost_per_test": 1.886966,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 43.049,
                "latency": 669.802,
                "stderr": 1.88,
                "cost_per_test": 0.187197,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "ant/ling-3.0-flash-2607": {
                "accuracy": 42.203,
                "latency": 163.407,
                "stderr": 0.99,
                "cost_per_test": 0.055985,
                "temperature": 0.6,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "poolside/laguna-m.1": {
                "accuracy": 39.252,
                "latency": 1201.969,
                "stderr": 3.343,
                "cost_per_test": 0.763505,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 37.241,
                "latency": 341.249,
                "stderr": 3.71,
                "cost_per_test": 0.866793,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "inception/mercury-2.5": {
                "accuracy": 31.353,
                "latency": 125.376,
                "stderr": 1.741,
                "cost_per_test": 0.109433,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Inception",
                "harness": null
            },
            "fireworks/nemotron-lightning-3p5-30b-a3b": {
                "accuracy": 28.531,
                "latency": 1201.053,
                "stderr": 1.084,
                "cost_per_test": 0.034263,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "poolside/laguna-xs.2": {
                "accuracy": 22.68,
                "latency": 527.313,
                "stderr": 2.817,
                "cost_per_test": 0.177571,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "cohere/command-a-plus-05-2026": {
                "accuracy": 8.899,
                "latency": 836.425,
                "stderr": 2.465,
                "cost_per_test": 1.899291,
                "temperature": 0.9,
                "top_p": 0.95,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            }
        },
        "comparables": {
            "google/gemini-3.8-flash": {
                "accuracy": 52.004,
                "latency": 276.762,
                "stderr": 0.158,
                "cost_per_test": 2.465927,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-4-argon": {
                "accuracy": 50.956,
                "latency": 1130.018,
                "stderr": 0.634,
                "cost_per_test": 5.506906,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.7-flash": {
                "accuracy": 50.283,
                "latency": 184.371,
                "stderr": 0.811,
                "cost_per_test": 1.834644,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "meta/muse_spark_1_2": {
                "accuracy": 50.239,
                "latency": 347.208,
                "stderr": 0.497,
                "cost_per_test": 0.90029,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "xiaomi/mimo-v2.6-flash": {
                "accuracy": 49.039,
                "latency": 463.807,
                "stderr": 1.223,
                "cost_per_test": 0.096154,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "google/gemini-3.5-flash": {
                "accuracy": 48.782,
                "latency": 380.721,
                "stderr": 0.311,
                "cost_per_test": 3.1206,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "mistralai/mistral-large-4": {
                "accuracy": 48.618,
                "latency": 3208.318,
                "stderr": 0.803,
                "cost_per_test": 1.570675,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "openai/gpt-5.6-luna": {
                "accuracy": 48.285,
                "latency": 1016.019,
                "stderr": 0.361,
                "cost_per_test": 0.369239,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3.6-flash": {
                "accuracy": 47.829,
                "latency": 509.046,
                "stderr": 0.796,
                "cost_per_test": 1.774881,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "kimi/kimi-k3": {
                "accuracy": 47.778,
                "latency": 342.039,
                "stderr": 0.237,
                "cost_per_test": 2.734443,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "ant/ling-3.0-flash-af-rc3": {
                "accuracy": 47.773,
                "latency": 455.22,
                "stderr": 0.593,
                "cost_per_test": 0.057351,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "anthropic/claude-opus-5": {
                "accuracy": 47.42,
                "latency": 686.987,
                "stderr": 0.532,
                "cost_per_test": 6.344139,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "meta/muse_spark_1_3": {
                "accuracy": 47.167,
                "latency": 394.811,
                "stderr": 0.448,
                "cost_per_test": 0.972175,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "meta/muse_spark_1_1": {
                "accuracy": 47.162,
                "latency": 515.058,
                "stderr": 0.42,
                "cost_per_test": 0.973188,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "anthropic/claude-sonnet-5-5": {
                "accuracy": 47.089,
                "latency": 2160.316,
                "stderr": 0.197,
                "cost_per_test": 7.64911,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.6-terra": {
                "accuracy": 47.032,
                "latency": 1680.653,
                "stderr": 6.422,
                "cost_per_test": 4.509465,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "stepfun/step-5-preview": {
                "accuracy": 46.414,
                "latency": 1127.621,
                "stderr": 0.56,
                "cost_per_test": 0.793346,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 1024000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Stepfun",
                "harness": null
            },
            "deepseek/deepseek-v4-pro-0813": {
                "accuracy": 46.193,
                "latency": 1436.908,
                "stderr": 0.996,
                "cost_per_test": 1.347639,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "deepseek/deepseek-v4.1-flash": {
                "accuracy": 46.054,
                "latency": 538.676,
                "stderr": 1.72,
                "cost_per_test": 0.282181,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "xiaomi/mimo-v2.6-pro": {
                "accuracy": 46.041,
                "latency": 687.272,
                "stderr": 0.878,
                "cost_per_test": 0.269754,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "anthropic/claude-fable-5-1": {
                "accuracy": 45.982,
                "latency": 1300.273,
                "stderr": 6.691,
                "cost_per_test": 10.771578,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "zai/glm-5.3-flash": {
                "accuracy": 45.936,
                "latency": 1156.287,
                "stderr": 6.635,
                "cost_per_test": 0.059522,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "anthropic/claude-sonnet-5": {
                "accuracy": 45.817,
                "latency": 1108.045,
                "stderr": 1.039,
                "cost_per_test": 3.091322,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-5-5": {
                "accuracy": 45.756,
                "latency": 2023.455,
                "stderr": 1.046,
                "cost_per_test": 10.619795,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "zai/glm-5.3": {
                "accuracy": 45.723,
                "latency": 1134.752,
                "stderr": 6.657,
                "cost_per_test": 1.269583,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "anthropic/claude-opus-4-7": {
                "accuracy": 45.556,
                "latency": 435.039,
                "stderr": 0.209,
                "cost_per_test": 6.088803,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-haiku-5-5": {
                "accuracy": 45.325,
                "latency": 1742.824,
                "stderr": 6.651,
                "cost_per_test": 1.955055,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.6-sol": {
                "accuracy": 45.284,
                "latency": 1405.294,
                "stderr": 0.98,
                "cost_per_test": 1.258215,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3.5-flash-lite": {
                "accuracy": 45.186,
                "latency": 433.722,
                "stderr": 0.602,
                "cost_per_test": 0.542906,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-4-8": {
                "accuracy": 44.853,
                "latency": 1048.58,
                "stderr": 1.122,
                "cost_per_test": 6.341675,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "zai/glm-5.2": {
                "accuracy": 44.78,
                "latency": 648.959,
                "stderr": 0.922,
                "cost_per_test": 1.029011,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "meta/muse_spark_1_3_max": {
                "accuracy": 44.467,
                "latency": 247.999,
                "stderr": 6.712,
                "cost_per_test": 0.980661,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "tencent/hy4-preview": {
                "accuracy": 44.37,
                "latency": 1705.997,
                "stderr": 1.369,
                "cost_per_test": 0.785409,
                "temperature": 0.9,
                "top_p": 1.0,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Tencent",
                "harness": null
            },
            "alibaba/qwen3.8-max": {
                "accuracy": 44.251,
                "latency": 1680.182,
                "stderr": 0.643,
                "cost_per_test": 1.636468,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "anthropic/claude-fable-5": {
                "accuracy": 43.986,
                "latency": 812.322,
                "stderr": 0.722,
                "cost_per_test": 10.480979,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 43.683,
                "latency": 734.01,
                "stderr": 0.904,
                "cost_per_test": 0.937845,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 43.653,
                "latency": 798.364,
                "stderr": 0.5,
                "cost_per_test": 5.684233,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "thinkingmachines/inkling": {
                "accuracy": 42.774,
                "latency": 1354.41,
                "stderr": 2.626,
                "cost_per_test": 1.284544,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "openai/gpt-6.1-sol": {
                "accuracy": 42.619,
                "latency": 1376.349,
                "stderr": 1.155,
                "cost_per_test": 2.399668,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3.7-max": {
                "accuracy": 42.324,
                "latency": 759.802,
                "stderr": 1.09,
                "cost_per_test": 2.038422,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "grok/grok-4.7": {
                "accuracy": 42.22,
                "latency": 1157.554,
                "stderr": 1.169,
                "cost_per_test": 3.693452,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "grok/grok-4.5": {
                "accuracy": 41.768,
                "latency": 377.919,
                "stderr": 1.428,
                "cost_per_test": 1.583537,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "fireworks/ember-1": {
                "accuracy": 41.672,
                "latency": 1359.884,
                "stderr": 6.31,
                "cost_per_test": 2.635755,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "minimax/MiniMax-M3": {
                "accuracy": 41.669,
                "latency": 577.327,
                "stderr": 3.912,
                "cost_per_test": 0.388805,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 512000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "zai/glm-5.1": {
                "accuracy": 41.557,
                "latency": 1058.07,
                "stderr": 0.234,
                "cost_per_test": 0.750335,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "grok/grok-4.6": {
                "accuracy": 41.229,
                "latency": 1166.009,
                "stderr": 0.17,
                "cost_per_test": 2.287188,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "deepseek/deepseek-v4-flash-0731": {
                "accuracy": 41.218,
                "latency": 645.509,
                "stderr": 0.973,
                "cost_per_test": 0.376029,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 41.069,
                "latency": 2307.494,
                "stderr": 0.105,
                "cost_per_test": 1.642419,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 41.003,
                "latency": 803.451,
                "stderr": 0.611,
                "cost_per_test": 3.196849,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-6-sol": {
                "accuracy": 40.833,
                "latency": 667.299,
                "stderr": 1.547,
                "cost_per_test": 2.950347,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 40.273,
                "latency": 2683.277,
                "stderr": 2.567,
                "cost_per_test": 1.509603,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "xiaomi/mimo-v2.5-pro": {
                "accuracy": 40.15,
                "latency": 538.572,
                "stderr": 1.84,
                "cost_per_test": 0.316099,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "openai/gpt-6-luna": {
                "accuracy": 39.772,
                "latency": 1256.828,
                "stderr": 1.108,
                "cost_per_test": 0.174364,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 39.632,
                "latency": 1805.003,
                "stderr": 1.323,
                "cost_per_test": 1.175258,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "alibaba/qwen3.8-27b": {
                "accuracy": 38.882,
                "latency": 822.187,
                "stderr": 6.17,
                "cost_per_test": 0.953854,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "alibaba/qwen3.7-plus": {
                "accuracy": 38.848,
                "latency": 1023.789,
                "stderr": 2.279,
                "cost_per_test": 0.486576,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "thinkingmachines/inkling-small": {
                "accuracy": 38.79,
                "latency": 1286.66,
                "stderr": 1.112,
                "cost_per_test": 0.300729,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 262000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 38.555,
                "latency": 279.59,
                "stderr": 1.771,
                "cost_per_test": 0.762386,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-6-astra": {
                "accuracy": 37.758,
                "latency": 1015.396,
                "stderr": 6.312,
                "cost_per_test": 9.441278,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "nvidia/nemotron-3-ultra-550b-a55b": {
                "accuracy": 37.181,
                "latency": 288.432,
                "stderr": 1.6,
                "cost_per_test": null,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Nvidia",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 37.173,
                "latency": 366.547,
                "stderr": 0.672,
                "cost_per_test": 0.207015,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 36.66,
                "latency": 268.379,
                "stderr": 2.083,
                "cost_per_test": 2.383117,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-4.3": {
                "accuracy": 35.857,
                "latency": 777.896,
                "stderr": 2.76,
                "cost_per_test": 1.111988,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 34.057,
                "latency": 61.889,
                "stderr": 0.849,
                "cost_per_test": 0.196515,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 32.13,
                "latency": 676.497,
                "stderr": 5.863,
                "cost_per_test": 0.327034,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "xiaomi/mimo-v2.5": {
                "accuracy": 31.672,
                "latency": 417.311,
                "stderr": 0.096,
                "cost_per_test": 0.114992,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "mistralai/mistral-medium-3.5": {
                "accuracy": 31.512,
                "latency": 1007.855,
                "stderr": 3.178,
                "cost_per_test": 2.097265,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 31.453,
                "latency": 203.88,
                "stderr": 1.208,
                "cost_per_test": 0.700359,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 30.188,
                "latency": 389.636,
                "stderr": 0.858,
                "cost_per_test": 0.862158,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 28.371,
                "latency": 994.669,
                "stderr": 0.702,
                "cost_per_test": 0.220717,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "poolside/laguna-m.1": {
                "accuracy": 28.012,
                "latency": 1763.392,
                "stderr": 3.455,
                "cost_per_test": 1.010782,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "ant/ling-3.0-flash-2607": {
                "accuracy": 27.435,
                "latency": 152.633,
                "stderr": 2.534,
                "cost_per_test": 0.046745,
                "temperature": 0.6,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "inception/mercury-2.5": {
                "accuracy": 25.496,
                "latency": 170.84,
                "stderr": 2.549,
                "cost_per_test": 0.115052,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Inception",
                "harness": null
            },
            "fireworks/nemotron-lightning-3p5-30b-a3b": {
                "accuracy": 24.282,
                "latency": 895.072,
                "stderr": 1.865,
                "cost_per_test": 0.038365,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "cohere/command-a-plus-05-2026": {
                "accuracy": 16.462,
                "latency": 681.636,
                "stderr": 1.001,
                "cost_per_test": 1.933787,
                "temperature": 0.9,
                "top_p": 0.95,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "poolside/laguna-xs.2": {
                "accuracy": 16.424,
                "latency": 505.067,
                "stderr": 1.515,
                "cost_per_test": 0.203936,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            }
        },
        "market_analysis": {
            "google/gemini-4-argon": {
                "accuracy": 79.705,
                "latency": 794.115,
                "stderr": 1.837,
                "cost_per_test": 3.782093,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.8-flash": {
                "accuracy": 78.5,
                "latency": 231.461,
                "stderr": 0.863,
                "cost_per_test": 1.504957,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.7-flash": {
                "accuracy": 74.513,
                "latency": 275.928,
                "stderr": 0.681,
                "cost_per_test": 1.055117,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-5": {
                "accuracy": 74.16,
                "latency": 501.718,
                "stderr": 1.471,
                "cost_per_test": 2.827226,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.5-flash": {
                "accuracy": 73.97,
                "latency": 231.63,
                "stderr": 1.911,
                "cost_per_test": 1.486992,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-sonnet-5-5": {
                "accuracy": 71.435,
                "latency": 1298.75,
                "stderr": 0.979,
                "cost_per_test": 4.192346,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "zai/glm-5.3-flash": {
                "accuracy": 70.417,
                "latency": 674.004,
                "stderr": 6.604,
                "cost_per_test": 0.030883,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "meta/muse_spark_1_3_max": {
                "accuracy": 69.948,
                "latency": 176.041,
                "stderr": 6.568,
                "cost_per_test": 0.577264,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "google/gemini-3.6-flash": {
                "accuracy": 69.831,
                "latency": 369.421,
                "stderr": 0.561,
                "cost_per_test": 1.023122,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-fable-5-1": {
                "accuracy": 69.544,
                "latency": 1067.022,
                "stderr": 6.553,
                "cost_per_test": 4.873172,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-5-5": {
                "accuracy": 69.422,
                "latency": 1290.592,
                "stderr": 1.964,
                "cost_per_test": 5.679642,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-fable-5": {
                "accuracy": 68.882,
                "latency": 474.571,
                "stderr": 2.144,
                "cost_per_test": 4.244023,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "zai/glm-5.3": {
                "accuracy": 67.396,
                "latency": 858.288,
                "stderr": 6.671,
                "cost_per_test": 0.736043,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "meta/muse_spark_1_1": {
                "accuracy": 67.233,
                "latency": 217.075,
                "stderr": 1.314,
                "cost_per_test": 0.458463,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "anthropic/claude-opus-4-8": {
                "accuracy": 67.219,
                "latency": 414.757,
                "stderr": 0.47,
                "cost_per_test": 2.017384,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "meta/muse_spark_1_2": {
                "accuracy": 66.201,
                "latency": 295.587,
                "stderr": 1.233,
                "cost_per_test": 0.549348,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "openai/gpt-6-astra": {
                "accuracy": 65.549,
                "latency": 529.935,
                "stderr": 6.784,
                "cost_per_test": 4.218887,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "meta/muse_spark_1_3": {
                "accuracy": 65.418,
                "latency": 366.7,
                "stderr": 2.192,
                "cost_per_test": 0.598154,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "openai/gpt-5.6-luna": {
                "accuracy": 64.896,
                "latency": 564.888,
                "stderr": 0.967,
                "cost_per_test": 0.207962,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-6.1-sol": {
                "accuracy": 64.467,
                "latency": 643.463,
                "stderr": 1.157,
                "cost_per_test": 0.979257,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.6-sol": {
                "accuracy": 64.163,
                "latency": 875.032,
                "stderr": 1.612,
                "cost_per_test": 1.828342,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-sonnet-5": {
                "accuracy": 64.075,
                "latency": 753.744,
                "stderr": 1.565,
                "cost_per_test": 1.18603,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3.8-max": {
                "accuracy": 63.724,
                "latency": 1028.593,
                "stderr": 0.549,
                "cost_per_test": 0.785001,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "xiaomi/mimo-v2.6-pro": {
                "accuracy": 63.454,
                "latency": 507.063,
                "stderr": 2.969,
                "cost_per_test": 0.110604,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "xiaomi/mimo-v2.6-flash": {
                "accuracy": 63.375,
                "latency": 376.708,
                "stderr": 0.288,
                "cost_per_test": 0.043635,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "deepseek/deepseek-v4.1-flash": {
                "accuracy": 63.361,
                "latency": 273.681,
                "stderr": 1.609,
                "cost_per_test": 0.096754,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "anthropic/claude-haiku-5-5": {
                "accuracy": 63.08,
                "latency": 1001.723,
                "stderr": 6.863,
                "cost_per_test": 0.684779,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "mistralai/mistral-large-4": {
                "accuracy": 62.984,
                "latency": 1021.978,
                "stderr": 0.999,
                "cost_per_test": 0.750243,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "anthropic/claude-opus-4-7": {
                "accuracy": 62.673,
                "latency": 253.644,
                "stderr": 1.978,
                "cost_per_test": 1.96134,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "ant/ling-3.0-flash-af-rc3": {
                "accuracy": 62.334,
                "latency": 302.105,
                "stderr": 2.6,
                "cost_per_test": 0.03336,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "grok/grok-4.6": {
                "accuracy": 62.247,
                "latency": 645.805,
                "stderr": 1.188,
                "cost_per_test": 1.072961,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "tencent/hy4-preview": {
                "accuracy": 62.143,
                "latency": 946.194,
                "stderr": 0.644,
                "cost_per_test": 0.307327,
                "temperature": 0.9,
                "top_p": 1.0,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Tencent",
                "harness": null
            },
            "kimi/kimi-k3": {
                "accuracy": 62.099,
                "latency": 271.418,
                "stderr": 1.608,
                "cost_per_test": 1.314267,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "grok/grok-4.7": {
                "accuracy": 61.832,
                "latency": 718.971,
                "stderr": 0.872,
                "cost_per_test": 2.0566,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "zai/glm-5.2": {
                "accuracy": 61.501,
                "latency": 405.12,
                "stderr": 1.829,
                "cost_per_test": 0.512704,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "openai/gpt-5.6-terra": {
                "accuracy": 60.623,
                "latency": 1074.484,
                "stderr": 6.895,
                "cost_per_test": 2.491446,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 60.288,
                "latency": 498.931,
                "stderr": 1.614,
                "cost_per_test": 2.721406,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "deepseek/deepseek-v4-pro-0813": {
                "accuracy": 60.268,
                "latency": 578.458,
                "stderr": 1.08,
                "cost_per_test": 0.44138,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "alibaba/qwen3.7-max": {
                "accuracy": 59.469,
                "latency": 446.147,
                "stderr": 2.338,
                "cost_per_test": 0.913863,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "minimax/MiniMax-M3": {
                "accuracy": 59.007,
                "latency": 369.203,
                "stderr": 0.331,
                "cost_per_test": 0.194569,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 512000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 59.004,
                "latency": 151.685,
                "stderr": 3.028,
                "cost_per_test": 0.296649,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 58.461,
                "latency": 448.902,
                "stderr": 2.367,
                "cost_per_test": 1.407303,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-6-sol": {
                "accuracy": 57.951,
                "latency": 412.402,
                "stderr": 0.633,
                "cost_per_test": 1.250149,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4.5": {
                "accuracy": 57.839,
                "latency": 190.111,
                "stderr": 0.742,
                "cost_per_test": 0.709931,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-3.5-flash-lite": {
                "accuracy": 57.022,
                "latency": 266.658,
                "stderr": 1.964,
                "cost_per_test": 0.245829,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 56.85,
                "latency": 1289.162,
                "stderr": 1.068,
                "cost_per_test": 0.809499,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "thinkingmachines/inkling": {
                "accuracy": 56.558,
                "latency": 767.016,
                "stderr": 1.544,
                "cost_per_test": 0.761876,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "fireworks/ember-1": {
                "accuracy": 56.549,
                "latency": 1354.716,
                "stderr": 7.073,
                "cost_per_test": 1.288145,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "stepfun/step-5-preview": {
                "accuracy": 55.815,
                "latency": 724.865,
                "stderr": 2.196,
                "cost_per_test": 0.380542,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 1024000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Stepfun",
                "harness": null
            },
            "deepseek/deepseek-v4-flash-0731": {
                "accuracy": 55.113,
                "latency": 325.97,
                "stderr": 2.229,
                "cost_per_test": 0.138833,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "openai/gpt-6-luna": {
                "accuracy": 53.283,
                "latency": 714.613,
                "stderr": 2.892,
                "cost_per_test": 0.104812,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 53.053,
                "latency": 1002.477,
                "stderr": 3.915,
                "cost_per_test": 0.577509,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 52.842,
                "latency": 134.603,
                "stderr": 1.717,
                "cost_per_test": 0.657507,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "zai/glm-5.1": {
                "accuracy": 51.861,
                "latency": 484.552,
                "stderr": 1.156,
                "cost_per_test": 0.352587,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 50.801,
                "latency": 889.762,
                "stderr": 1.742,
                "cost_per_test": 0.389785,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "grok/grok-4.3": {
                "accuracy": 50.245,
                "latency": 371.126,
                "stderr": 0.753,
                "cost_per_test": 0.459991,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "alibaba/qwen3.8-27b": {
                "accuracy": 49.572,
                "latency": 459.738,
                "stderr": 7.006,
                "cost_per_test": 0.64433,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "thinkingmachines/inkling-small": {
                "accuracy": 49.303,
                "latency": 662.045,
                "stderr": 1.442,
                "cost_per_test": 0.135783,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 262000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "xiaomi/mimo-v2.5-pro": {
                "accuracy": 48.971,
                "latency": 328.815,
                "stderr": 0.391,
                "cost_per_test": 0.094477,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "xiaomi/mimo-v2.5": {
                "accuracy": 48.567,
                "latency": 246.009,
                "stderr": 1.659,
                "cost_per_test": 0.03703,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "alibaba/qwen3.7-plus": {
                "accuracy": 48.47,
                "latency": 506.411,
                "stderr": 1.552,
                "cost_per_test": 0.148204,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 47.324,
                "latency": 355.663,
                "stderr": 1.192,
                "cost_per_test": 0.373151,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 46.461,
                "latency": 237.62,
                "stderr": 3.042,
                "cost_per_test": 0.088647,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 42.851,
                "latency": 420.475,
                "stderr": 6.565,
                "cost_per_test": 0.137681,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "nvidia/nemotron-3-ultra-550b-a55b": {
                "accuracy": 41.27,
                "latency": 208.881,
                "stderr": 1.624,
                "cost_per_test": null,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Nvidia",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 40.674,
                "latency": 109.817,
                "stderr": 1.737,
                "cost_per_test": 0.301453,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "ant/ling-3.0-flash-2607": {
                "accuracy": 37.301,
                "latency": 103.788,
                "stderr": 1.797,
                "cost_per_test": 0.02867,
                "temperature": 0.6,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "mistralai/mistral-medium-3.5": {
                "accuracy": 37.023,
                "latency": 560.062,
                "stderr": 1.316,
                "cost_per_test": 1.648409,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "poolside/laguna-m.1": {
                "accuracy": 35.579,
                "latency": 686.193,
                "stderr": 2.141,
                "cost_per_test": 0.433226,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 34.189,
                "latency": 39.442,
                "stderr": 1.174,
                "cost_per_test": 0.094503,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 33.392,
                "latency": 404.99,
                "stderr": 1.419,
                "cost_per_test": 0.119445,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 33.329,
                "latency": 209.24,
                "stderr": 1.501,
                "cost_per_test": 0.446844,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "inception/mercury-2.5": {
                "accuracy": 26.624,
                "latency": 71.922,
                "stderr": 0.4,
                "cost_per_test": 0.082934,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Inception",
                "harness": null
            },
            "fireworks/nemotron-lightning-3p5-30b-a3b": {
                "accuracy": 21.392,
                "latency": 911.287,
                "stderr": 3.601,
                "cost_per_test": 0.01966,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "poolside/laguna-xs.2": {
                "accuracy": 20.433,
                "latency": 330.888,
                "stderr": 2.368,
                "cost_per_test": 0.126566,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "cohere/command-a-plus-05-2026": {
                "accuracy": 11.963,
                "latency": 327.572,
                "stderr": 0.937,
                "cost_per_test": 1.179474,
                "temperature": 0.9,
                "top_p": 0.95,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            }
        }
    }
}