{
    "metadata": {
        "benchmark": "Terminal-Bench 2.1",
        "slug": "terminal-bench-2-1",
        "description": "State-of-the-art set of difficult terminal-based tasks",
        "benchmark_id": "terminal_bench_2_1",
        "family": "terminal_bench",
        "version": "2.1",
        "updated": "2026-09-28",
        "dataset_type": "public",
        "industry": "coding",
        "tasks": {
            "overall": "Overall",
            "easy": "Easy",
            "medium": "Medium",
            "hard": "Hard"
        },
        "models": [
            "alibaba/qwen3.6-plus",
            "alibaba/qwen3.7-max",
            "alibaba/qwen3.7-plus",
            "alibaba/qwen3.8-27b",
            "alibaba/qwen3.8-max",
            "ant/ling-3.0-flash-2607",
            "ant/ling-3.0-flash-af-rc3",
            "anthropic/claude-fable-5",
            "anthropic/claude-fable-5-1",
            "anthropic/claude-haiku-4-5-20251001-thinking",
            "anthropic/claude-opus-4-7",
            "anthropic/claude-opus-4-8",
            "anthropic/claude-opus-4-8-claude-code",
            "anthropic/claude-opus-5",
            "anthropic/claude-opus-5-5",
            "anthropic/claude-sonnet-4-6",
            "anthropic/claude-sonnet-5",
            "anthropic/claude-sonnet-5-5",
            "cohere/command-a-plus-05-2026",
            "cursor/composer-2.5",
            "deepseek/deepseek-v4-flash-0731",
            "deepseek/deepseek-v4-pro",
            "deepseek/deepseek-v4-pro-0813",
            "deepseek/deepseek-v4.1-flash",
            "fireworks/nemotron-lightning-3p5-30b-a3b",
            "google/barium-bb",
            "google/gemini-3-flash-preview",
            "google/gemini-3.1-flash-lite-preview",
            "google/gemini-3.1-pro-preview",
            "google/gemini-3.5-flash",
            "google/gemini-3.5-flash-lite",
            "google/gemini-3.6-flash",
            "google/gemini-3.7-flash",
            "google/gemini-3.8-flash",
            "grok/grok-4.20-0309-reasoning",
            "grok/grok-4.3",
            "grok/grok-4.5",
            "grok/grok-4.6",
            "grok/grok-4.7",
            "inception/mercury-2.5",
            "kimi/kimi-k2.5-thinking",
            "kimi/kimi-k2.6",
            "kimi/kimi-k2.7-code",
            "kimi/kimi-k3",
            "meta/muse_spark_1_1",
            "meta/muse_spark_1_2",
            "meta/muse_spark_1_3",
            "meta/muse_spark_1_3_max",
            "minimax/MiniMax-M2.7",
            "minimax/MiniMax-M3",
            "mistralai/mistral-medium-3.5",
            "nvidia/nemotron-3-ultra-550b-a55b",
            "openai/gpt-5.4-mini-2026-03-17",
            "openai/gpt-5.4-nano-2026-03-17",
            "openai/gpt-5.5",
            "openai/gpt-5.5-codex",
            "openai/gpt-5.5-factory",
            "openai/gpt-5.6-luna",
            "openai/gpt-5.6-sol",
            "openai/gpt-5.6-terra",
            "openai/gpt-6-astra",
            "openai/gpt-6-luna",
            "openai/gpt-6-sol",
            "poolside/laguna-m.1",
            "poolside/laguna-xs.2",
            "tencent/hy4-preview",
            "thinkingmachines/inkling",
            "thinkingmachines/inkling-small",
            "xiaomi/mimo-v2.5",
            "xiaomi/mimo-v2.5-pro",
            "xiaomi/mimo-v2.6-flash",
            "xiaomi/mimo-v2.6-pro",
            "zai/glm-5.1",
            "zai/glm-5.2",
            "zai/glm-5.3",
            "zai/glm-5.3-flash"
        ],
        "partners": [],
        "showBadge": true,
        "visible": true,
        "use_cost_per_test": true,
        "harness": "Terminus 2",
        "runner": "external",
        "mode": "agentic",
        "archived": false,
        "partner": false,
        "total_models": 76
    },
    "tasks": {
        "overall": {
            "anthropic/claude-opus-5-5": {
                "accuracy": 87.64,
                "latency": 383.079,
                "stderr": 1.716,
                "cost_per_test": 0.496033,
                "token_totals": {
                    "input_tokens": 22792900,
                    "output_tokens": 1198947,
                    "reasoning_tokens": 351772,
                    "cache_read_tokens": 20604830,
                    "cache_write_tokens": 2181727
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-6-astra": {
                "accuracy": 87.266,
                "latency": 274.44,
                "stderr": 0.375,
                "cost_per_test": 1.339493,
                "token_totals": {
                    "input_tokens": 24715799,
                    "output_tokens": 1309769,
                    "reasoning_tokens": 764679,
                    "cache_read_tokens": 21492398,
                    "cache_write_tokens": 0
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.6-sol": {
                "accuracy": 85.768,
                "latency": 372.819,
                "stderr": 1.35,
                "cost_per_test": 1.023319,
                "token_totals": {
                    "input_tokens": 29114624,
                    "output_tokens": 1324119,
                    "reasoning_tokens": 721900,
                    "cache_read_tokens": 20938069,
                    "cache_write_tokens": 0
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-fable-5-1": {
                "accuracy": 85.019,
                "latency": 500.355,
                "stderr": 0.991,
                "cost_per_test": 3.074796,
                "token_totals": {
                    "input_tokens": 21892452,
                    "output_tokens": 965107,
                    "reasoning_tokens": 472161,
                    "cache_read_tokens": 19902668,
                    "cache_write_tokens": 1985257
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-5": {
                "accuracy": 84.644,
                "latency": 752.932,
                "stderr": 0.991,
                "cost_per_test": 0.886936,
                "token_totals": {
                    "input_tokens": 35206371,
                    "output_tokens": 1733729,
                    "reasoning_tokens": 1113936,
                    "cache_read_tokens": 32076924,
                    "cache_write_tokens": 3126721
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-6-sol": {
                "accuracy": 83.146,
                "latency": 298.876,
                "stderr": 1.297,
                "cost_per_test": 0.373024,
                "token_totals": {
                    "input_tokens": 53182967,
                    "output_tokens": 1568343,
                    "reasoning_tokens": 998593,
                    "cache_read_tokens": 50178777,
                    "cache_write_tokens": 2943219
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3.8-flash": {
                "accuracy": 81.273,
                "latency": 554.797,
                "stderr": 0.375,
                "cost_per_test": 1.544972,
                "token_totals": {
                    "input_tokens": 325795812,
                    "output_tokens": 4741834,
                    "reasoning_tokens": 2933371,
                    "cache_read_tokens": 286485175,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "kimi/kimi-k3": {
                "accuracy": 80.899,
                "latency": 608.575,
                "stderr": 0.649,
                "cost_per_test": 0.341034,
                "token_totals": {
                    "input_tokens": 20494904,
                    "output_tokens": 1244660,
                    "reasoning_tokens": 843947,
                    "cache_read_tokens": 0,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "anthropic/claude-fable-5": {
                "accuracy": 80.524,
                "latency": 504.537,
                "stderr": 1.35,
                "cost_per_test": 1.429025,
                "token_totals": {
                    "input_tokens": 21092499,
                    "output_tokens": 1498447,
                    "reasoning_tokens": 0,
                    "cache_read_tokens": 18355304,
                    "cache_write_tokens": 2613456
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.6-luna": {
                "accuracy": 79.026,
                "latency": 355.751,
                "stderr": 0.991,
                "cost_per_test": 0.053767,
                "token_totals": {
                    "input_tokens": 67560156,
                    "output_tokens": 2052883,
                    "reasoning_tokens": 1203057,
                    "cache_read_tokens": 62168050,
                    "cache_write_tokens": 0
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "meta/muse_spark_1_3_max": {
                "accuracy": 79.026,
                "latency": 731.78,
                "stderr": 0.991,
                "cost_per_test": 0.59671,
                "token_totals": {
                    "input_tokens": 198539978,
                    "output_tokens": 3621191,
                    "reasoning_tokens": 2589082,
                    "cache_read_tokens": 191325278,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "grok/grok-4.6": {
                "accuracy": 78.277,
                "latency": 642.908,
                "stderr": 2.085,
                "cost_per_test": 0.453642,
                "token_totals": {
                    "input_tokens": 37989503,
                    "output_tokens": 2301698,
                    "reasoning_tokens": 1606577,
                    "cache_read_tokens": 32943360,
                    "cache_write_tokens": 0
                },
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-3.7-flash": {
                "accuracy": 77.528,
                "latency": 492.202,
                "stderr": 0.649,
                "cost_per_test": 1.022219,
                "token_totals": {
                    "input_tokens": 195502339,
                    "output_tokens": 3101625,
                    "reasoning_tokens": 1363982,
                    "cache_read_tokens": 167065334,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.6-terra": {
                "accuracy": 77.528,
                "latency": 519.739,
                "stderr": 2.247,
                "cost_per_test": 0.472552,
                "token_totals": {
                    "input_tokens": 42753712,
                    "output_tokens": 2043151,
                    "reasoning_tokens": 1345891,
                    "cache_read_tokens": 37760047,
                    "cache_write_tokens": 0
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 76.404,
                "latency": 427.412,
                "stderr": 0.649,
                "cost_per_test": 0.741883,
                "token_totals": {
                    "input_tokens": 16386293,
                    "output_tokens": 1376238,
                    "reasoning_tokens": 851034,
                    "cache_read_tokens": 12709120,
                    "cache_write_tokens": 0
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "xiaomi/mimo-v2.6-flash": {
                "accuracy": 76.404,
                "latency": 809.733,
                "stderr": 1.716,
                "cost_per_test": 0.02332,
                "token_totals": {
                    "input_tokens": 77103048,
                    "output_tokens": 3904513,
                    "reasoning_tokens": 3173128,
                    "cache_read_tokens": 71517419,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "deepseek/deepseek-v4.1-flash": {
                "accuracy": 74.532,
                "latency": 604.506,
                "stderr": 1.633,
                "cost_per_test": 0.098877,
                "token_totals": {
                    "input_tokens": 56873906,
                    "output_tokens": 6392867,
                    "reasoning_tokens": 5656496,
                    "cache_read_tokens": 54195840,
                    "cache_write_tokens": 0
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "anthropic/claude-sonnet-5": {
                "accuracy": 74.532,
                "latency": 635.043,
                "stderr": 2.085,
                "cost_per_test": 0.534595,
                "token_totals": {
                    "input_tokens": 74219799,
                    "output_tokens": 2321306,
                    "reasoning_tokens": 0,
                    "cache_read_tokens": 70230688,
                    "cache_write_tokens": 3987009
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.5-flash": {
                "accuracy": 74.157,
                "latency": 373.223,
                "stderr": 1.124,
                "cost_per_test": 0.868179,
                "token_totals": {
                    "input_tokens": 135043681,
                    "output_tokens": 2323941,
                    "reasoning_tokens": 196148,
                    "cache_read_tokens": 108305986,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.6-flash": {
                "accuracy": 73.783,
                "latency": 552.153,
                "stderr": 1.633,
                "cost_per_test": 1.174404,
                "token_totals": {
                    "input_tokens": 224181261,
                    "output_tokens": 3381533,
                    "reasoning_tokens": 1807265,
                    "cache_read_tokens": 190452944,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-4.7": {
                "accuracy": 73.408,
                "latency": 797.952,
                "stderr": 1.498,
                "cost_per_test": 1.076601,
                "token_totals": {
                    "input_tokens": 98794136,
                    "output_tokens": 2621813,
                    "reasoning_tokens": 1801692,
                    "cache_read_tokens": 78833280,
                    "cache_write_tokens": 0
                },
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-6-luna": {
                "accuracy": 73.034,
                "latency": 425.094,
                "stderr": 1.716,
                "cost_per_test": 0.029222,
                "token_totals": {
                    "input_tokens": 68771981,
                    "output_tokens": 2862029,
                    "reasoning_tokens": 2313628,
                    "cache_read_tokens": 64567254,
                    "cache_write_tokens": 4144652
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "meta/muse_spark_1_3": {
                "accuracy": 72.285,
                "latency": 857.636,
                "stderr": 0.375,
                "cost_per_test": 0.663249,
                "token_totals": {
                    "input_tokens": 153551412,
                    "output_tokens": 3070064,
                    "reasoning_tokens": 2095052,
                    "cache_read_tokens": 132688964,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "anthropic/claude-opus-4-8": {
                "accuracy": 71.91,
                "latency": 929.902,
                "stderr": 0.649,
                "cost_per_test": 2.409772,
                "token_totals": {
                    "input_tokens": 84371822,
                    "output_tokens": 5125339,
                    "reasoning_tokens": 0,
                    "cache_read_tokens": 76658876,
                    "cache_write_tokens": 7553615
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "zai/glm-5.3": {
                "accuracy": 71.536,
                "latency": 786.604,
                "stderr": 1.982,
                "cost_per_test": 0.30672,
                "token_totals": {
                    "input_tokens": 46365317,
                    "output_tokens": 2447978,
                    "reasoning_tokens": 1787588,
                    "cache_read_tokens": 42442496,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 70.787,
                "latency": 410.249,
                "stderr": 1.124,
                "cost_per_test": 0.582677,
                "token_totals": {
                    "input_tokens": 33939400,
                    "output_tokens": 2667201,
                    "reasoning_tokens": 2048217,
                    "cache_read_tokens": 26933702,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-4-8-claude-code": {
                "accuracy": 69.663,
                "latency": 359.967,
                "stderr": 4.901,
                "cost_per_test": 0.968098,
                "token_totals": {
                    "input_tokens": 62415811,
                    "output_tokens": 1520645,
                    "reasoning_tokens": 0,
                    "cache_read_tokens": 0,
                    "cache_write_tokens": 0
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": "Claude Code"
            },
            "meta/muse_spark_1_2": {
                "accuracy": 69.663,
                "latency": 879.273,
                "stderr": 0.649,
                "cost_per_test": 0.499472,
                "token_totals": {
                    "input_tokens": 131957427,
                    "output_tokens": 3332616,
                    "reasoning_tokens": 2186108,
                    "cache_read_tokens": 122415847,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "meta/muse_spark_1_1": {
                "accuracy": 69.288,
                "latency": 624.772,
                "stderr": 0.991,
                "cost_per_test": 0.417396,
                "token_totals": {
                    "input_tokens": 138955979,
                    "output_tokens": 3124998,
                    "reasoning_tokens": 1933296,
                    "cache_read_tokens": 136207264,
                    "cache_write_tokens": 0
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "anthropic/claude-opus-4-7": {
                "accuracy": 68.539,
                "latency": 816.587,
                "stderr": 1.297,
                "cost_per_test": 1.977087,
                "token_totals": {
                    "input_tokens": 109753697,
                    "output_tokens": 3484336,
                    "reasoning_tokens": 0,
                    "cache_read_tokens": 103807239,
                    "cache_write_tokens": 5773113
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-4.5": {
                "accuracy": 67.79,
                "latency": 624.415,
                "stderr": 4.416,
                "cost_per_test": 0.283943,
                "token_totals": {
                    "input_tokens": 26039357,
                    "output_tokens": 1388617,
                    "reasoning_tokens": 651797,
                    "cache_read_tokens": 23426304,
                    "cache_write_tokens": 0
                },
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "zai/glm-5.2": {
                "accuracy": 67.79,
                "latency": 820.506,
                "stderr": 0.991,
                "cost_per_test": 0.425184,
                "token_totals": {
                    "input_tokens": 58211167,
                    "output_tokens": 3186166,
                    "reasoning_tokens": 2550689,
                    "cache_read_tokens": 50590720,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 48000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "xiaomi/mimo-v2.6-pro": {
                "accuracy": 67.79,
                "latency": 939.123,
                "stderr": 1.873,
                "cost_per_test": 0.072656,
                "token_totals": {
                    "input_tokens": 62497955,
                    "output_tokens": 4342791,
                    "reasoning_tokens": 3692566,
                    "cache_read_tokens": 56788224,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "alibaba/qwen3.8-max": {
                "accuracy": 67.416,
                "latency": 890.917,
                "stderr": 0.649,
                "cost_per_test": 0.448084,
                "token_totals": {
                    "input_tokens": 43506113,
                    "output_tokens": 2658371,
                    "reasoning_tokens": 2141241,
                    "cache_read_tokens": 36047403,
                    "cache_write_tokens": 0
                },
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "deepseek/deepseek-v4-flash-0731": {
                "accuracy": 67.041,
                "latency": 683.03,
                "stderr": 1.35,
                "cost_per_test": 0.021068,
                "token_totals": {
                    "input_tokens": 54370624,
                    "output_tokens": 4785981,
                    "reasoning_tokens": 3945320,
                    "cache_read_tokens": 51580757,
                    "cache_write_tokens": 0
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": false,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "kimi/kimi-k2.7-code": {
                "accuracy": 67.041,
                "latency": 784.159,
                "stderr": 0.375,
                "cost_per_test": 0.256503,
                "token_totals": {
                    "input_tokens": 66026998,
                    "output_tokens": 1942653,
                    "reasoning_tokens": 0,
                    "cache_read_tokens": 62720352,
                    "cache_write_tokens": 0
                },
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "zai/glm-5.3-flash": {
                "accuracy": 62.921,
                "latency": 965.737,
                "stderr": 2.247,
                "cost_per_test": 0.028403,
                "token_totals": {
                    "input_tokens": 87337009,
                    "output_tokens": 3334054,
                    "reasoning_tokens": 0,
                    "cache_read_tokens": 80932267,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "alibaba/qwen3.7-max": {
                "accuracy": 61.049,
                "latency": 935.196,
                "stderr": 0.375,
                "cost_per_test": 0.33233,
                "token_totals": {
                    "input_tokens": 20525184,
                    "output_tokens": 1673824,
                    "reasoning_tokens": 1260207,
                    "cache_read_tokens": 15239680,
                    "cache_write_tokens": 0
                },
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "xiaomi/mimo-v2.5": {
                "accuracy": 60.674,
                "latency": 775.239,
                "stderr": 5.207,
                "cost_per_test": 0.021658,
                "token_totals": {
                    "input_tokens": 86304814,
                    "output_tokens": 3550628,
                    "reasoning_tokens": 2582392,
                    "cache_read_tokens": 81263296,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "cursor/composer-2.5": {
                "accuracy": 58.427,
                "latency": 0.0,
                "stderr": 5.254,
                "cost_per_test": null,
                "token_totals": {
                    "input_tokens": 0,
                    "output_tokens": 0,
                    "reasoning_tokens": 0,
                    "cache_read_tokens": 0,
                    "cache_write_tokens": 0
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 200000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cursor",
                "harness": "Cursor CLI"
            },
            "alibaba/qwen3.8-27b": {
                "accuracy": 58.427,
                "latency": 892.585,
                "stderr": 5.254,
                "cost_per_test": 0.677989,
                "token_totals": {
                    "input_tokens": 95606478,
                    "output_tokens": 4179262,
                    "reasoning_tokens": 3317358,
                    "cache_read_tokens": 0,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "google/barium-bb": {
                "accuracy": 57.678,
                "latency": 841.802,
                "stderr": 2.456,
                "cost_per_test": null,
                "token_totals": {
                    "input_tokens": 24915374,
                    "output_tokens": 4239234,
                    "reasoning_tokens": 3409824,
                    "cache_read_tokens": 15655837,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.5-codex": {
                "accuracy": 57.303,
                "latency": 202.469,
                "stderr": 5.273,
                "cost_per_test": 0.645217,
                "token_totals": {
                    "input_tokens": 49121192,
                    "output_tokens": 613091,
                    "reasoning_tokens": 0,
                    "cache_read_tokens": 0,
                    "cache_write_tokens": 0
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": "Codex"
            },
            "openai/gpt-5.5-factory": {
                "accuracy": 57.303,
                "latency": 389.265,
                "stderr": 5.273,
                "cost_per_test": 0.751531,
                "token_totals": {
                    "input_tokens": 39439811,
                    "output_tokens": 744918,
                    "reasoning_tokens": 0,
                    "cache_read_tokens": 0,
                    "cache_write_tokens": 0
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": "Factory"
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 57.303,
                "latency": 704.225,
                "stderr": 0.649,
                "cost_per_test": 0.572392,
                "token_totals": {
                    "input_tokens": 52390316,
                    "output_tokens": 1600704,
                    "reasoning_tokens": 0,
                    "cache_read_tokens": 49128739,
                    "cache_write_tokens": 3211949
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": false,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "xiaomi/mimo-v2.5-pro": {
                "accuracy": 57.303,
                "latency": 866.253,
                "stderr": 5.273,
                "cost_per_test": 0.043613,
                "token_totals": {
                    "input_tokens": 55124187,
                    "output_tokens": 2331687,
                    "reasoning_tokens": 1538667,
                    "cache_read_tokens": 51288960,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "zai/glm-5.1": {
                "accuracy": 56.929,
                "latency": 940.943,
                "stderr": 1.35,
                "cost_per_test": 0.232386,
                "token_totals": {
                    "input_tokens": 50906500,
                    "output_tokens": 1815010,
                    "reasoning_tokens": 1071648,
                    "cache_read_tokens": 45040256,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "thinkingmachines/inkling-small": {
                "accuracy": 55.056,
                "latency": 603.689,
                "stderr": 2.339,
                "cost_per_test": 0.127848,
                "token_totals": {
                    "input_tokens": 112525802,
                    "output_tokens": 2949484,
                    "reasoning_tokens": 0,
                    "cache_read_tokens": 107994240,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 262000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "tencent/hy4-preview": {
                "accuracy": 55.056,
                "latency": 1109.129,
                "stderr": 3.433,
                "cost_per_test": 0.140288,
                "token_totals": {
                    "input_tokens": 58874316,
                    "output_tokens": 2524449,
                    "reasoning_tokens": 1360245,
                    "cache_read_tokens": 54651691,
                    "cache_write_tokens": 0
                },
                "temperature": 0.9,
                "top_p": 1.0,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Tencent",
                "harness": null
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 54.682,
                "latency": 673.036,
                "stderr": 4.556,
                "cost_per_test": 0.286739,
                "token_totals": {
                    "input_tokens": 20716925,
                    "output_tokens": 4555260,
                    "reasoning_tokens": 3735861,
                    "cache_read_tokens": 15580075,
                    "cache_write_tokens": 0
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "deepseek/deepseek-v4-pro-0813": {
                "accuracy": 54.682,
                "latency": 862.374,
                "stderr": 1.498,
                "cost_per_test": 0.199883,
                "token_totals": {
                    "input_tokens": 41799035,
                    "output_tokens": 3293030,
                    "reasoning_tokens": 2648603,
                    "cache_read_tokens": 39518421,
                    "cache_write_tokens": 0
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 53.933,
                "latency": 376.158,
                "stderr": 1.297,
                "cost_per_test": 0.154312,
                "token_totals": {
                    "input_tokens": 40724231,
                    "output_tokens": 2528658,
                    "reasoning_tokens": 1486431,
                    "cache_read_tokens": 31587323,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 53.558,
                "latency": 932.358,
                "stderr": 2.085,
                "cost_per_test": 0.138449,
                "token_totals": {
                    "input_tokens": 21929461,
                    "output_tokens": 1568451,
                    "reasoning_tokens": 0,
                    "cache_read_tokens": 18714965,
                    "cache_write_tokens": 0
                },
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "minimax/MiniMax-M3": {
                "accuracy": 53.558,
                "latency": 1023.416,
                "stderr": 0.749,
                "cost_per_test": 0.205973,
                "token_totals": {
                    "input_tokens": 64485322,
                    "output_tokens": 2119689,
                    "reasoning_tokens": 0,
                    "cache_read_tokens": 53014239,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 512000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 53.184,
                "latency": 778.109,
                "stderr": 1.633,
                "cost_per_test": 0.201902,
                "token_totals": {
                    "input_tokens": 22196424,
                    "output_tokens": 2290370,
                    "reasoning_tokens": 1654262,
                    "cache_read_tokens": 0,
                    "cache_write_tokens": 0
                },
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "alibaba/qwen3.7-plus": {
                "accuracy": 52.809,
                "latency": 654.983,
                "stderr": 0.649,
                "cost_per_test": 0.095849,
                "token_totals": {
                    "input_tokens": 53627914,
                    "output_tokens": 1867100,
                    "reasoning_tokens": 1116365,
                    "cache_read_tokens": 44188800,
                    "cache_write_tokens": 0
                },
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "nvidia/nemotron-3-ultra-550b-a55b": {
                "accuracy": 50.936,
                "latency": 584.116,
                "stderr": 2.925,
                "cost_per_test": null,
                "token_totals": {
                    "input_tokens": 111289804,
                    "output_tokens": 2704899,
                    "reasoning_tokens": 0,
                    "cache_read_tokens": 0,
                    "cache_write_tokens": 0
                },
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Nvidia",
                "harness": null
            },
            "google/gemini-3.5-flash-lite": {
                "accuracy": 50.187,
                "latency": 416.383,
                "stderr": 0.991,
                "cost_per_test": 0.337052,
                "token_totals": {
                    "input_tokens": 247145811,
                    "output_tokens": 2141665,
                    "reasoning_tokens": 1125438,
                    "cache_read_tokens": 183334301,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "ant/ling-3.0-flash-2607": {
                "accuracy": 50.187,
                "latency": 593.826,
                "stderr": 2.456,
                "cost_per_test": 0.067399,
                "token_totals": {
                    "input_tokens": 210224480,
                    "output_tokens": 3744590,
                    "reasoning_tokens": 0,
                    "cache_read_tokens": 194621013,
                    "cache_write_tokens": 0
                },
                "temperature": 0.6,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "ant/ling-3.0-flash-af-rc3": {
                "accuracy": 50.187,
                "latency": 833.528,
                "stderr": 1.35,
                "cost_per_test": 0.043393,
                "token_totals": {
                    "input_tokens": 136334405,
                    "output_tokens": 7151114,
                    "reasoning_tokens": 0,
                    "cache_read_tokens": 118584064,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 50.187,
                "latency": 1078.158,
                "stderr": 1.498,
                "cost_per_test": 0.239191,
                "token_totals": {
                    "input_tokens": 36128386,
                    "output_tokens": 3648724,
                    "reasoning_tokens": 2947326,
                    "cache_read_tokens": 34026923,
                    "cache_write_tokens": 0
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": false,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 48.689,
                "latency": 812.638,
                "stderr": 3.066,
                "cost_per_test": 0.066012,
                "token_totals": {
                    "input_tokens": 90437019,
                    "output_tokens": 1897074,
                    "reasoning_tokens": 0,
                    "cache_read_tokens": 87265378,
                    "cache_write_tokens": 388109
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "thinkingmachines/inkling": {
                "accuracy": 47.566,
                "latency": 771.551,
                "stderr": 0.375,
                "cost_per_test": 0.630691,
                "token_totals": {
                    "input_tokens": 226227937,
                    "output_tokens": 3179128,
                    "reasoning_tokens": 0,
                    "cache_read_tokens": 220634923,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 44.195,
                "latency": 305.726,
                "stderr": 0.991,
                "cost_per_test": 0.407362,
                "token_totals": {
                    "input_tokens": 28275678,
                    "output_tokens": 2268536,
                    "reasoning_tokens": 1952049,
                    "cache_read_tokens": 18837397,
                    "cache_write_tokens": 0
                },
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 43.82,
                "latency": 649.489,
                "stderr": 1.946,
                "cost_per_test": 0.363822,
                "token_totals": {
                    "input_tokens": 84235092,
                    "output_tokens": 4076150,
                    "reasoning_tokens": 0,
                    "cache_read_tokens": 80999717,
                    "cache_write_tokens": 2656167
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-4.3": {
                "accuracy": 41.948,
                "latency": 470.713,
                "stderr": 2.622,
                "cost_per_test": 0.982203,
                "token_totals": {
                    "input_tokens": 232260469,
                    "output_tokens": 2597032,
                    "reasoning_tokens": 1420634,
                    "cache_read_tokens": 220796992,
                    "cache_write_tokens": 0
                },
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 41.948,
                "latency": 1051.079,
                "stderr": 0.749,
                "cost_per_test": 0.085844,
                "token_totals": {
                    "input_tokens": 34513305,
                    "output_tokens": 1049061,
                    "reasoning_tokens": 355219,
                    "cache_read_tokens": 32430174,
                    "cache_write_tokens": 0
                },
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 41.573,
                "latency": 757.485,
                "stderr": 2.973,
                "cost_per_test": 0.072321,
                "token_totals": {
                    "input_tokens": 30849406,
                    "output_tokens": 3569635,
                    "reasoning_tokens": 2418488,
                    "cache_read_tokens": 23307435,
                    "cache_write_tokens": 0
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "mistralai/mistral-medium-3.5": {
                "accuracy": 38.951,
                "latency": 989.169,
                "stderr": 2.701,
                "cost_per_test": 1.190984,
                "token_totals": {
                    "input_tokens": 55841015,
                    "output_tokens": 2964803,
                    "reasoning_tokens": 0,
                    "cache_read_tokens": 0,
                    "cache_write_tokens": 0
                },
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "inception/mercury-2.5": {
                "accuracy": 34.457,
                "latency": 688.923,
                "stderr": 0.749,
                "cost_per_test": 0.214987,
                "token_totals": {
                    "input_tokens": 158620249,
                    "output_tokens": 2324434,
                    "reasoning_tokens": 820396,
                    "cache_read_tokens": 79630802,
                    "cache_write_tokens": 0
                },
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Inception",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 34.082,
                "latency": 278.342,
                "stderr": 0.991,
                "cost_per_test": 0.084894,
                "token_totals": {
                    "input_tokens": 36882464,
                    "output_tokens": 2980609,
                    "reasoning_tokens": 2461322,
                    "cache_read_tokens": 27270935,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "poolside/laguna-m.1": {
                "accuracy": 34.082,
                "latency": 1209.608,
                "stderr": 1.633,
                "cost_per_test": 0.300575,
                "token_totals": {
                    "input_tokens": 223144546,
                    "output_tokens": 2120066,
                    "reasoning_tokens": 495317,
                    "cache_read_tokens": 187257533,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "poolside/laguna-xs.2": {
                "accuracy": 25.843,
                "latency": 1218.578,
                "stderr": 2.973,
                "cost_per_test": 0.253714,
                "token_totals": {
                    "input_tokens": 291488884,
                    "output_tokens": 2546703,
                    "reasoning_tokens": 621785,
                    "cache_read_tokens": 141552645,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "cohere/command-a-plus-05-2026": {
                "accuracy": 17.603,
                "latency": 0.0,
                "stderr": 0.375,
                "cost_per_test": null,
                "token_totals": {
                    "input_tokens": 0,
                    "output_tokens": 0,
                    "reasoning_tokens": 0,
                    "cache_read_tokens": 0,
                    "cache_write_tokens": 0
                },
                "temperature": 0.9,
                "top_p": 0.95,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "fireworks/nemotron-lightning-3p5-30b-a3b": {
                "accuracy": 10.861,
                "latency": 1028.327,
                "stderr": 2.622,
                "cost_per_test": 0.014296,
                "token_totals": {
                    "input_tokens": 73183797,
                    "output_tokens": 2194003,
                    "reasoning_tokens": 0,
                    "cache_read_tokens": 70641781,
                    "cache_write_tokens": 0
                },
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "anthropic/claude-sonnet-5-5": {
                "accuracy": 83.146,
                "latency": 465.828,
                "stderr": 1.716,
                "cost_per_test": 0.625494,
                "token_totals": {
                    "input_tokens": 47223436,
                    "output_tokens": 3585507,
                    "reasoning_tokens": 379696,
                    "cache_read_tokens": 43257785,
                    "cache_write_tokens": 3960025
                },
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            }
        },
        "easy": {
            "cursor/composer-2.5": {
                "accuracy": 100.0,
                "latency": 0.0,
                "stderr": 0.0,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 200000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cursor",
                "harness": "Cursor CLI"
            },
            "openai/gpt-6-astra": {
                "accuracy": 100.0,
                "latency": 6106.294,
                "stderr": 0.0,
                "cost_per_test": 29.803716,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-6-sol": {
                "accuracy": 100.0,
                "latency": 6649.999,
                "stderr": 0.0,
                "cost_per_test": 8.299794,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-4-8-claude-code": {
                "accuracy": 100.0,
                "latency": 8009.262,
                "stderr": 0.0,
                "cost_per_test": 21.540183,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": "Claude Code"
            },
            "google/gemini-3.5-flash": {
                "accuracy": 100.0,
                "latency": 8304.209,
                "stderr": 0.0,
                "cost_per_test": 19.316978,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-5-5": {
                "accuracy": 100.0,
                "latency": 8523.514,
                "stderr": 0.0,
                "cost_per_test": 0.30277,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-fable-5-1": {
                "accuracy": 100.0,
                "latency": 11132.898,
                "stderr": 0.0,
                "cost_per_test": 1.741133,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-fable-5": {
                "accuracy": 100.0,
                "latency": 11225.953,
                "stderr": 0.0,
                "cost_per_test": 9.538742,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.6-terra": {
                "accuracy": 100.0,
                "latency": 11564.187,
                "stderr": 0.0,
                "cost_per_test": 0.248207,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3.6-flash": {
                "accuracy": 100.0,
                "latency": 12285.415,
                "stderr": 0.0,
                "cost_per_test": 26.130479,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.8-flash": {
                "accuracy": 100.0,
                "latency": 12344.24,
                "stderr": 0.0,
                "cost_per_test": 0.931088,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-sonnet-5": {
                "accuracy": 100.0,
                "latency": 14129.714,
                "stderr": 0.0,
                "cost_per_test": 0.269427,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 100.0,
                "latency": 15669.003,
                "stderr": 0.0,
                "cost_per_test": 12.735719,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": false,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "meta/muse_spark_1_3_max": {
                "accuracy": 100.0,
                "latency": 16282.108,
                "stderr": 0.0,
                "cost_per_test": 13.276807,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "anthropic/claude-opus-5": {
                "accuracy": 100.0,
                "latency": 16752.727,
                "stderr": 0.0,
                "cost_per_test": 0.44962,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "xiaomi/mimo-v2.5": {
                "accuracy": 100.0,
                "latency": 17249.061,
                "stderr": 0.0,
                "cost_per_test": 0.481881,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "zai/glm-5.2": {
                "accuracy": 100.0,
                "latency": 18256.253,
                "stderr": 0.0,
                "cost_per_test": 9.460336,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 48000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "anthropic/claude-opus-4-8": {
                "accuracy": 100.0,
                "latency": 20690.317,
                "stderr": 0.0,
                "cost_per_test": 32.170449,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.6-sol": {
                "accuracy": 91.667,
                "latency": 8295.214,
                "stderr": 8.333,
                "cost_per_test": 22.768842,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 91.667,
                "latency": 9128.048,
                "stderr": 8.333,
                "cost_per_test": 12.964574,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-6-luna": {
                "accuracy": 91.667,
                "latency": 9458.34,
                "stderr": 8.333,
                "cost_per_test": 0.650194,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3.7-flash": {
                "accuracy": 91.667,
                "latency": 10951.504,
                "stderr": 8.333,
                "cost_per_test": 22.744373,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "nvidia/nemotron-3-ultra-550b-a55b": {
                "accuracy": 91.667,
                "latency": 12996.574,
                "stderr": 8.333,
                "cost_per_test": null,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Nvidia",
                "harness": null
            },
            "thinkingmachines/inkling-small": {
                "accuracy": 91.667,
                "latency": 13432.073,
                "stderr": 8.333,
                "cost_per_test": 0.066748,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 262000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "deepseek/deepseek-v4.1-flash": {
                "accuracy": 91.667,
                "latency": 13450.264,
                "stderr": 8.333,
                "cost_per_test": 2.200009,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "kimi/kimi-k3": {
                "accuracy": 91.667,
                "latency": 13540.8,
                "stderr": 8.333,
                "cost_per_test": 0.142639,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "grok/grok-4.5": {
                "accuracy": 91.667,
                "latency": 13893.236,
                "stderr": 8.333,
                "cost_per_test": 6.317739,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "deepseek/deepseek-v4-flash-0731": {
                "accuracy": 91.667,
                "latency": 15197.407,
                "stderr": 8.333,
                "cost_per_test": 0.46877,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": false,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "thinkingmachines/inkling": {
                "accuracy": 91.667,
                "latency": 17167.019,
                "stderr": 8.333,
                "cost_per_test": 0.193793,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "kimi/kimi-k2.7-code": {
                "accuracy": 91.667,
                "latency": 17447.537,
                "stderr": 8.333,
                "cost_per_test": 5.707198,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "zai/glm-5.3": {
                "accuracy": 91.667,
                "latency": 17501.946,
                "stderr": 8.333,
                "cost_per_test": 0.102681,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "grok/grok-4.7": {
                "accuracy": 91.667,
                "latency": 17754.441,
                "stderr": 8.333,
                "cost_per_test": 23.954366,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-opus-4-7": {
                "accuracy": 91.667,
                "latency": 18169.067,
                "stderr": 8.333,
                "cost_per_test": 43.990177,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "ant/ling-3.0-flash-af-rc3": {
                "accuracy": 91.667,
                "latency": 18546.005,
                "stderr": 8.333,
                "cost_per_test": 0.017196,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "meta/muse_spark_1_2": {
                "accuracy": 91.667,
                "latency": 19563.827,
                "stderr": 8.333,
                "cost_per_test": 11.113242,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "zai/glm-5.3-flash": {
                "accuracy": 91.667,
                "latency": 21487.644,
                "stderr": 8.333,
                "cost_per_test": 0.631963,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 83.333,
                "latency": 6802.405,
                "stderr": 8.333,
                "cost_per_test": 9.063814,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-5.6-luna": {
                "accuracy": 83.333,
                "latency": 7915.469,
                "stderr": 16.667,
                "cost_per_test": 0.020477,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 83.333,
                "latency": 9509.922,
                "stderr": 8.333,
                "cost_per_test": 16.506894,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "ant/ling-3.0-flash-2607": {
                "accuracy": 83.333,
                "latency": 13212.628,
                "stderr": 8.333,
                "cost_per_test": 0.025539,
                "temperature": 0.6,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 83.333,
                "latency": 14451.125,
                "stderr": 8.333,
                "cost_per_test": 8.095034,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 83.333,
                "latency": 16854.051,
                "stderr": 16.667,
                "cost_per_test": 1.609147,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "xiaomi/mimo-v2.6-flash": {
                "accuracy": 83.333,
                "latency": 18016.562,
                "stderr": 8.333,
                "cost_per_test": 0.518875,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 83.333,
                "latency": 20744.962,
                "stderr": 8.333,
                "cost_per_test": 3.080492,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "alibaba/qwen3.7-max": {
                "accuracy": 83.333,
                "latency": 20808.115,
                "stderr": 8.333,
                "cost_per_test": 7.39434,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "zai/glm-5.1": {
                "accuracy": 83.333,
                "latency": 20935.973,
                "stderr": 8.333,
                "cost_per_test": 5.170582,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "minimax/MiniMax-M3": {
                "accuracy": 83.333,
                "latency": 22771.001,
                "stderr": 8.333,
                "cost_per_test": 4.582903,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 512000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "openai/gpt-5.5-codex": {
                "accuracy": 75.0,
                "latency": 4504.93,
                "stderr": 25.0,
                "cost_per_test": 14.356089,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": "Codex"
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 75.0,
                "latency": 8369.51,
                "stderr": 14.434,
                "cost_per_test": 3.433448,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.5-factory": {
                "accuracy": 75.0,
                "latency": 8661.154,
                "stderr": 25.0,
                "cost_per_test": 16.721574,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": "Factory"
            },
            "grok/grok-4.3": {
                "accuracy": 75.0,
                "latency": 10473.371,
                "stderr": 0.0,
                "cost_per_test": 21.85401,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "meta/muse_spark_1_1": {
                "accuracy": 75.0,
                "latency": 13901.169,
                "stderr": 0.0,
                "cost_per_test": 0.168085,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "grok/grok-4.6": {
                "accuracy": 75.0,
                "latency": 14304.713,
                "stderr": 0.0,
                "cost_per_test": 0.25745,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "alibaba/qwen3.7-plus": {
                "accuracy": 75.0,
                "latency": 14573.382,
                "stderr": 14.434,
                "cost_per_test": 2.13264,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 75.0,
                "latency": 14975.056,
                "stderr": 0.0,
                "cost_per_test": 6.379954,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 75.0,
                "latency": 17312.922,
                "stderr": 0.0,
                "cost_per_test": 4.49233,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 75.0,
                "latency": 18081.196,
                "stderr": 0.0,
                "cost_per_test": 1.468763,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "google/barium-bb": {
                "accuracy": 75.0,
                "latency": 18730.092,
                "stderr": 0.0,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "meta/muse_spark_1_3": {
                "accuracy": 75.0,
                "latency": 19082.404,
                "stderr": 0.0,
                "cost_per_test": 14.757295,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "deepseek/deepseek-v4-pro-0813": {
                "accuracy": 75.0,
                "latency": 19187.816,
                "stderr": 14.434,
                "cost_per_test": 0.09918,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "xiaomi/mimo-v2.5-pro": {
                "accuracy": 75.0,
                "latency": 19274.134,
                "stderr": 25.0,
                "cost_per_test": 0.970383,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "alibaba/qwen3.8-max": {
                "accuracy": 75.0,
                "latency": 19822.896,
                "stderr": 0.0,
                "cost_per_test": 9.969874,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "xiaomi/mimo-v2.6-pro": {
                "accuracy": 75.0,
                "latency": 20895.493,
                "stderr": 0.0,
                "cost_per_test": 1.6166,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 75.0,
                "latency": 23386.5,
                "stderr": 0.0,
                "cost_per_test": 1.91002,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "poolside/laguna-m.1": {
                "accuracy": 75.0,
                "latency": 26913.783,
                "stderr": 0.0,
                "cost_per_test": 6.687796,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 66.667,
                "latency": 6193.118,
                "stderr": 8.333,
                "cost_per_test": 1.888892,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.5-flash-lite": {
                "accuracy": 66.667,
                "latency": 9264.533,
                "stderr": 8.333,
                "cost_per_test": 7.499411,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "inception/mercury-2.5": {
                "accuracy": 66.667,
                "latency": 15328.538,
                "stderr": 8.333,
                "cost_per_test": 4.783458,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Inception",
                "harness": null
            },
            "mistralai/mistral-medium-3.5": {
                "accuracy": 66.667,
                "latency": 22009.013,
                "stderr": 8.333,
                "cost_per_test": 26.499387,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 66.667,
                "latency": 23989.013,
                "stderr": 8.333,
                "cost_per_test": 5.322002,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": false,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "tencent/hy4-preview": {
                "accuracy": 58.333,
                "latency": 24678.121,
                "stderr": 16.667,
                "cost_per_test": 0.060606,
                "temperature": 0.9,
                "top_p": 1.0,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Tencent",
                "harness": null
            },
            "poolside/laguna-xs.2": {
                "accuracy": 58.333,
                "latency": 27113.353,
                "stderr": 8.333,
                "cost_per_test": 5.645149,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "cohere/command-a-plus-05-2026": {
                "accuracy": 50.0,
                "latency": 0.0,
                "stderr": 0.0,
                "cost_per_test": null,
                "temperature": 0.9,
                "top_p": 0.95,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "alibaba/qwen3.8-27b": {
                "accuracy": 50.0,
                "latency": 19860.021,
                "stderr": 28.868,
                "cost_per_test": 0.47166,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "fireworks/nemotron-lightning-3p5-30b-a3b": {
                "accuracy": 41.667,
                "latency": 22880.27,
                "stderr": 16.667,
                "cost_per_test": 0.31808,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "anthropic/claude-sonnet-5-5": {
                "accuracy": 91.667,
                "latency": 10364.662,
                "stderr": 8.333,
                "cost_per_test": 0.49001,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            }
        },
        "medium": {
            "anthropic/claude-opus-5-5": {
                "accuracy": 90.303,
                "latency": 619.892,
                "stderr": 1.603,
                "cost_per_test": 0.368791,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-6-astra": {
                "accuracy": 86.061,
                "latency": 444.094,
                "stderr": 0.606,
                "cost_per_test": 2.167543,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.6-sol": {
                "accuracy": 86.061,
                "latency": 603.288,
                "stderr": 2.185,
                "cost_per_test": 1.655916,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-fable-5-1": {
                "accuracy": 85.455,
                "latency": 809.665,
                "stderr": 1.05,
                "cost_per_test": 2.649676,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-fable-5": {
                "accuracy": 84.848,
                "latency": 816.433,
                "stderr": 0.606,
                "cost_per_test": 0.693727,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.8-flash": {
                "accuracy": 84.848,
                "latency": 897.763,
                "stderr": 0.606,
                "cost_per_test": 1.355444,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-6-sol": {
                "accuracy": 84.242,
                "latency": 483.636,
                "stderr": 1.603,
                "cost_per_test": 0.603621,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.6-luna": {
                "accuracy": 83.636,
                "latency": 575.67,
                "stderr": 1.05,
                "cost_per_test": 0.039659,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3.7-flash": {
                "accuracy": 83.636,
                "latency": 796.473,
                "stderr": 1.818,
                "cost_per_test": 1.654136,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "kimi/kimi-k3": {
                "accuracy": 83.636,
                "latency": 984.785,
                "stderr": 1.818,
                "cost_per_test": 0.215305,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "grok/grok-4.6": {
                "accuracy": 83.636,
                "latency": 1040.343,
                "stderr": 1.818,
                "cost_per_test": 0.363143,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-opus-5": {
                "accuracy": 83.636,
                "latency": 1218.38,
                "stderr": 1.05,
                "cost_per_test": 0.722162,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "deepseek/deepseek-v4.1-flash": {
                "accuracy": 83.03,
                "latency": 978.201,
                "stderr": 1.212,
                "cost_per_test": 0.160001,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "anthropic/claude-opus-4-7": {
                "accuracy": 82.424,
                "latency": 1321.387,
                "stderr": 0.606,
                "cost_per_test": 3.199286,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-sonnet-5": {
                "accuracy": 81.818,
                "latency": 1027.616,
                "stderr": 2.777,
                "cost_per_test": 0.369097,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 81.212,
                "latency": 691.631,
                "stderr": 0.606,
                "cost_per_test": 1.200501,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "meta/muse_spark_1_3_max": {
                "accuracy": 81.212,
                "latency": 1184.153,
                "stderr": 1.212,
                "cost_per_test": 0.965586,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "google/gemini-3.5-flash": {
                "accuracy": 80.0,
                "latency": 603.942,
                "stderr": 1.05,
                "cost_per_test": 1.404871,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "zai/glm-5.3": {
                "accuracy": 80.0,
                "latency": 1272.869,
                "stderr": 1.818,
                "cost_per_test": 0.211885,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "openai/gpt-5.6-terra": {
                "accuracy": 79.394,
                "latency": 841.032,
                "stderr": 1.603,
                "cost_per_test": 0.367085,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "deepseek/deepseek-v4-flash-0731": {
                "accuracy": 79.394,
                "latency": 1105.266,
                "stderr": 1.603,
                "cost_per_test": 0.034092,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": false,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "meta/muse_spark_1_1": {
                "accuracy": 78.182,
                "latency": 1010.994,
                "stderr": 1.05,
                "cost_per_test": 0.312504,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "meta/muse_spark_1_3": {
                "accuracy": 78.182,
                "latency": 1387.811,
                "stderr": 1.05,
                "cost_per_test": 1.073258,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "meta/muse_spark_1_2": {
                "accuracy": 78.182,
                "latency": 1422.824,
                "stderr": 1.818,
                "cost_per_test": 0.808236,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "google/gemini-3.6-flash": {
                "accuracy": 76.97,
                "latency": 893.485,
                "stderr": 1.603,
                "cost_per_test": 1.900398,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "kimi/kimi-k2.7-code": {
                "accuracy": 76.97,
                "latency": 1268.912,
                "stderr": 3.207,
                "cost_per_test": 0.415069,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "anthropic/claude-opus-4-8": {
                "accuracy": 76.364,
                "latency": 1504.75,
                "stderr": 1.05,
                "cost_per_test": 2.339669,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-6-luna": {
                "accuracy": 75.758,
                "latency": 687.879,
                "stderr": 2.424,
                "cost_per_test": 0.047287,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 75.152,
                "latency": 663.858,
                "stderr": 1.212,
                "cost_per_test": 0.942878,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-4.7": {
                "accuracy": 75.152,
                "latency": 1291.232,
                "stderr": 1.212,
                "cost_per_test": 1.742136,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "xiaomi/mimo-v2.6-flash": {
                "accuracy": 75.152,
                "latency": 1310.295,
                "stderr": 1.603,
                "cost_per_test": 0.037736,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "zai/glm-5.2": {
                "accuracy": 74.545,
                "latency": 1327.727,
                "stderr": 1.05,
                "cost_per_test": 0.688024,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 48000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "alibaba/qwen3.8-max": {
                "accuracy": 74.545,
                "latency": 1441.665,
                "stderr": 1.05,
                "cost_per_test": 0.725082,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "zai/glm-5.3-flash": {
                "accuracy": 73.939,
                "latency": 1562.738,
                "stderr": 3.207,
                "cost_per_test": 0.045961,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "grok/grok-4.5": {
                "accuracy": 72.121,
                "latency": 1010.417,
                "stderr": 5.781,
                "cost_per_test": 0.459472,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-opus-4-8-claude-code": {
                "accuracy": 70.909,
                "latency": 582.492,
                "stderr": 6.181,
                "cost_per_test": 1.566559,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": "Claude Code"
            },
            "xiaomi/mimo-v2.5": {
                "accuracy": 70.909,
                "latency": 1254.477,
                "stderr": 6.181,
                "cost_per_test": 0.035046,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "xiaomi/mimo-v2.6-pro": {
                "accuracy": 69.697,
                "latency": 1519.672,
                "stderr": 4.242,
                "cost_per_test": 0.117571,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "alibaba/qwen3.7-max": {
                "accuracy": 68.485,
                "latency": 1513.317,
                "stderr": 1.212,
                "cost_per_test": 0.53777,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "alibaba/qwen3.7-plus": {
                "accuracy": 67.273,
                "latency": 1059.882,
                "stderr": 1.818,
                "cost_per_test": 0.155101,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "deepseek/deepseek-v4-pro-0813": {
                "accuracy": 67.273,
                "latency": 1395.478,
                "stderr": 2.777,
                "cost_per_test": 0.159222,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "alibaba/qwen3.8-27b": {
                "accuracy": 67.273,
                "latency": 1444.365,
                "stderr": 6.385,
                "cost_per_test": 0.547552,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "thinkingmachines/inkling-small": {
                "accuracy": 66.667,
                "latency": 976.878,
                "stderr": 2.185,
                "cost_per_test": 0.098682,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 262000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 66.667,
                "latency": 1139.564,
                "stderr": 1.603,
                "cost_per_test": 0.926234,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": false,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 66.667,
                "latency": 1508.725,
                "stderr": 3.687,
                "cost_per_test": 0.224036,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "cursor/composer-2.5": {
                "accuracy": 65.455,
                "latency": 0.0,
                "stderr": 6.471,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 200000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cursor",
                "harness": "Cursor CLI"
            },
            "openai/gpt-5.5-factory": {
                "accuracy": 65.455,
                "latency": 629.902,
                "stderr": 6.471,
                "cost_per_test": 1.216114,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": "Factory"
            },
            "xiaomi/mimo-v2.5-pro": {
                "accuracy": 65.455,
                "latency": 1401.755,
                "stderr": 6.471,
                "cost_per_test": 0.070573,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "zai/glm-5.1": {
                "accuracy": 65.455,
                "latency": 1522.616,
                "stderr": 1.05,
                "cost_per_test": 0.376042,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 64.242,
                "latency": 1089.095,
                "stderr": 3.687,
                "cost_per_test": 0.463997,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "minimax/MiniMax-M3": {
                "accuracy": 64.242,
                "latency": 1656.073,
                "stderr": 1.603,
                "cost_per_test": 0.333302,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 512000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "openai/gpt-5.5-codex": {
                "accuracy": 63.636,
                "latency": 327.631,
                "stderr": 6.546,
                "cost_per_test": 1.044079,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": "Codex"
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 63.636,
                "latency": 1259.122,
                "stderr": 2.099,
                "cost_per_test": 0.326715,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 63.03,
                "latency": 1314.996,
                "stderr": 2.642,
                "cost_per_test": 0.106819,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "tencent/hy4-preview": {
                "accuracy": 63.03,
                "latency": 1794.772,
                "stderr": 3.207,
                "cost_per_test": 0.105198,
                "temperature": 0.9,
                "top_p": 1.0,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Tencent",
                "harness": null
            },
            "nvidia/nemotron-3-ultra-550b-a55b": {
                "accuracy": 62.424,
                "latency": 945.205,
                "stderr": 2.424,
                "cost_per_test": null,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Nvidia",
                "harness": null
            },
            "ant/ling-3.0-flash-2607": {
                "accuracy": 62.424,
                "latency": 960.918,
                "stderr": 2.424,
                "cost_per_test": 0.056094,
                "temperature": 0.6,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 60.0,
                "latency": 608.692,
                "stderr": 2.777,
                "cost_per_test": 0.249705,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.5-flash-lite": {
                "accuracy": 59.394,
                "latency": 673.784,
                "stderr": 1.603,
                "cost_per_test": 0.545412,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 59.394,
                "latency": 1744.656,
                "stderr": 1.603,
                "cost_per_test": 0.387055,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": false,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "google/barium-bb": {
                "accuracy": 58.788,
                "latency": 1362.189,
                "stderr": 2.185,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "thinkingmachines/inkling": {
                "accuracy": 57.576,
                "latency": 1248.51,
                "stderr": 1.212,
                "cost_per_test": 0.489649,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "ant/ling-3.0-flash-af-rc3": {
                "accuracy": 57.576,
                "latency": 1348.8,
                "stderr": 2.185,
                "cost_per_test": 0.032348,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 56.364,
                "latency": 494.72,
                "stderr": 1.05,
                "cost_per_test": 0.659186,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "grok/grok-4.3": {
                "accuracy": 54.545,
                "latency": 761.7,
                "stderr": 3.785,
                "cost_per_test": 1.589383,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 52.727,
                "latency": 1050.991,
                "stderr": 3.785,
                "cost_per_test": 0.58873,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 52.121,
                "latency": 1700.836,
                "stderr": 1.603,
                "cost_per_test": 0.138911,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "mistralai/mistral-medium-3.5": {
                "accuracy": 49.697,
                "latency": 1600.656,
                "stderr": 4.242,
                "cost_per_test": 1.927228,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 47.273,
                "latency": 1225.749,
                "stderr": 1.05,
                "cost_per_test": 0.117029,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "inception/mercury-2.5": {
                "accuracy": 44.242,
                "latency": 1114.803,
                "stderr": 0.606,
                "cost_per_test": 0.347888,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Inception",
                "harness": null
            },
            "poolside/laguna-m.1": {
                "accuracy": 42.424,
                "latency": 1957.366,
                "stderr": 2.642,
                "cost_per_test": 0.486385,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 41.818,
                "latency": 450.409,
                "stderr": 1.05,
                "cost_per_test": 0.137374,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "poolside/laguna-xs.2": {
                "accuracy": 31.515,
                "latency": 1971.88,
                "stderr": 3.207,
                "cost_per_test": 0.410556,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "cohere/command-a-plus-05-2026": {
                "accuracy": 23.03,
                "latency": 0.0,
                "stderr": 0.606,
                "cost_per_test": null,
                "temperature": 0.9,
                "top_p": 0.95,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "fireworks/nemotron-lightning-3p5-30b-a3b": {
                "accuracy": 13.939,
                "latency": 1664.02,
                "stderr": 3.374,
                "cost_per_test": 0.023133,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "anthropic/claude-sonnet-5-5": {
                "accuracy": 86.061,
                "latency": 753.794,
                "stderr": 1.212,
                "cost_per_test": 0.550492,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            }
        },
        "hard": {
            "openai/gpt-6-astra": {
                "accuracy": 87.778,
                "latency": 814.173,
                "stderr": 2.222,
                "cost_per_test": 3.973829,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.6-sol": {
                "accuracy": 84.444,
                "latency": 1106.029,
                "stderr": 2.94,
                "cost_per_test": 3.035846,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-5": {
                "accuracy": 84.444,
                "latency": 2233.697,
                "stderr": 1.111,
                "cost_per_test": 1.247332,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-fable-5-1": {
                "accuracy": 82.222,
                "latency": 1484.386,
                "stderr": 1.111,
                "cost_per_test": 4.032005,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-5-5": {
                "accuracy": 81.111,
                "latency": 1136.469,
                "stderr": 2.94,
                "cost_per_test": 0.746597,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-6-sol": {
                "accuracy": 78.889,
                "latency": 886.667,
                "stderr": 2.94,
                "cost_per_test": 1.106639,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "xiaomi/mimo-v2.6-flash": {
                "accuracy": 77.778,
                "latency": 2402.208,
                "stderr": 2.94,
                "cost_per_test": 0.069183,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "kimi/kimi-k3": {
                "accuracy": 74.444,
                "latency": 1805.44,
                "stderr": 6.186,
                "cost_per_test": 0.597989,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "google/gemini-3.8-flash": {
                "accuracy": 72.222,
                "latency": 1645.899,
                "stderr": 2.222,
                "cost_per_test": 1.97429,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "meta/muse_spark_1_3_max": {
                "accuracy": 72.222,
                "latency": 2170.948,
                "stderr": 1.111,
                "cost_per_test": 1.770241,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "openai/gpt-5.6-terra": {
                "accuracy": 71.111,
                "latency": 1541.892,
                "stderr": 4.006,
                "cost_per_test": 0.695821,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.6-luna": {
                "accuracy": 70.0,
                "latency": 1055.396,
                "stderr": 1.925,
                "cost_per_test": 0.084069,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-fable-5": {
                "accuracy": 70.0,
                "latency": 1496.794,
                "stderr": 3.333,
                "cost_per_test": 1.271832,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-4.6": {
                "accuracy": 68.889,
                "latency": 1907.295,
                "stderr": 2.94,
                "cost_per_test": 0.645717,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "grok/grok-4.7": {
                "accuracy": 67.778,
                "latency": 2367.259,
                "stderr": 2.94,
                "cost_per_test": 3.193915,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 66.667,
                "latency": 1267.99,
                "stderr": 1.925,
                "cost_per_test": 2.200919,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-6-luna": {
                "accuracy": 65.556,
                "latency": 1261.112,
                "stderr": 2.94,
                "cost_per_test": 0.086693,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3.7-flash": {
                "accuracy": 64.444,
                "latency": 1460.201,
                "stderr": 2.222,
                "cost_per_test": 3.032583,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.6-flash": {
                "accuracy": 64.444,
                "latency": 1638.055,
                "stderr": 2.222,
                "cost_per_test": 3.484064,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-4-8-claude-code": {
                "accuracy": 63.333,
                "latency": 1067.902,
                "stderr": 8.949,
                "cost_per_test": 2.872024,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": "Claude Code"
            },
            "xiaomi/mimo-v2.6-pro": {
                "accuracy": 63.333,
                "latency": 2786.066,
                "stderr": 3.333,
                "cost_per_test": 0.215547,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "meta/muse_spark_1_3": {
                "accuracy": 61.111,
                "latency": 2544.321,
                "stderr": 1.111,
                "cost_per_test": 1.967639,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "google/gemini-3.5-flash": {
                "accuracy": 60.0,
                "latency": 1107.228,
                "stderr": 1.925,
                "cost_per_test": 2.575597,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 60.0,
                "latency": 1217.073,
                "stderr": 1.925,
                "cost_per_test": 1.72861,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-4-8": {
                "accuracy": 60.0,
                "latency": 2758.709,
                "stderr": 1.925,
                "cost_per_test": 4.289393,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-sonnet-5": {
                "accuracy": 57.778,
                "latency": 1883.962,
                "stderr": 2.222,
                "cost_per_test": 0.873362,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "deepseek/deepseek-v4.1-flash": {
                "accuracy": 56.667,
                "latency": 1793.369,
                "stderr": 1.925,
                "cost_per_test": 0.293335,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "grok/grok-4.5": {
                "accuracy": 56.667,
                "latency": 1852.431,
                "stderr": 1.925,
                "cost_per_test": 0.842365,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "zai/glm-5.3": {
                "accuracy": 53.333,
                "latency": 2333.593,
                "stderr": 3.333,
                "cost_per_test": 0.50779,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "google/barium-bb": {
                "accuracy": 53.333,
                "latency": 2497.346,
                "stderr": 3.333,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3.8-max": {
                "accuracy": 53.333,
                "latency": 2643.053,
                "stderr": 1.925,
                "cost_per_test": 1.329316,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "meta/muse_spark_1_1": {
                "accuracy": 52.222,
                "latency": 1853.489,
                "stderr": 2.94,
                "cost_per_test": 0.642939,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "zai/glm-5.2": {
                "accuracy": 51.111,
                "latency": 2434.167,
                "stderr": 2.222,
                "cost_per_test": 1.261378,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 48000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "meta/muse_spark_1_2": {
                "accuracy": 51.111,
                "latency": 2608.51,
                "stderr": 1.111,
                "cost_per_test": 1.481766,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "kimi/kimi-k2.7-code": {
                "accuracy": 45.556,
                "latency": 2326.338,
                "stderr": 4.006,
                "cost_per_test": 0.76096,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "alibaba/qwen3.7-max": {
                "accuracy": 44.444,
                "latency": 2774.415,
                "stderr": 1.111,
                "cost_per_test": 0.985912,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/gpt-5.5-codex": {
                "accuracy": 43.333,
                "latency": 600.657,
                "stderr": 9.202,
                "cost_per_test": 1.914145,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": "Codex"
            },
            "alibaba/qwen3.8-27b": {
                "accuracy": 43.333,
                "latency": 2648.003,
                "stderr": 9.202,
                "cost_per_test": 0.944634,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "deepseek/deepseek-v4-flash-0731": {
                "accuracy": 41.111,
                "latency": 2026.321,
                "stderr": 2.222,
                "cost_per_test": 0.062503,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": false,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "cursor/composer-2.5": {
                "accuracy": 40.0,
                "latency": 0.0,
                "stderr": 9.097,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 200000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cursor",
                "harness": "Cursor CLI"
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 40.0,
                "latency": 1115.935,
                "stderr": 3.333,
                "cost_per_test": 0.457793,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.5-factory": {
                "accuracy": 40.0,
                "latency": 1154.821,
                "stderr": 9.097,
                "cost_per_test": 2.229543,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": "Factory"
            },
            "anthropic/claude-opus-4-7": {
                "accuracy": 40.0,
                "latency": 2422.542,
                "stderr": 5.092,
                "cost_per_test": 5.865357,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "xiaomi/mimo-v2.5-pro": {
                "accuracy": 40.0,
                "latency": 2569.885,
                "stderr": 9.097,
                "cost_per_test": 0.129384,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "tencent/hy4-preview": {
                "accuracy": 40.0,
                "latency": 3290.416,
                "stderr": 3.849,
                "cost_per_test": 0.212905,
                "temperature": 0.9,
                "top_p": 1.0,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Tencent",
                "harness": null
            },
            "zai/glm-5.3-flash": {
                "accuracy": 38.889,
                "latency": 2865.019,
                "stderr": 2.94,
                "cost_per_test": 0.084262,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "zai/glm-5.1": {
                "accuracy": 37.778,
                "latency": 2791.463,
                "stderr": 1.111,
                "cost_per_test": 0.689411,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "xiaomi/mimo-v2.5": {
                "accuracy": 36.667,
                "latency": 2299.875,
                "stderr": 8.949,
                "cost_per_test": 0.064251,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 34.444,
                "latency": 1996.674,
                "stderr": 6.759,
                "cost_per_test": 0.85066,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 34.444,
                "latency": 2089.2,
                "stderr": 2.94,
                "cost_per_test": 1.698096,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": false,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.5-flash-lite": {
                "accuracy": 31.111,
                "latency": 1235.271,
                "stderr": 1.111,
                "cost_per_test": 0.999921,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 31.111,
                "latency": 2308.39,
                "stderr": 1.111,
                "cost_per_test": 0.598977,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "ant/ling-3.0-flash-af-rc3": {
                "accuracy": 31.111,
                "latency": 2472.801,
                "stderr": 1.111,
                "cost_per_test": 0.066401,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 31.111,
                "latency": 3198.535,
                "stderr": 2.222,
                "cost_per_test": 0.7096,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": false,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "minimax/MiniMax-M3": {
                "accuracy": 30.0,
                "latency": 3036.133,
                "stderr": 1.925,
                "cost_per_test": 0.611054,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 512000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "thinkingmachines/inkling-small": {
                "accuracy": 28.889,
                "latency": 1790.943,
                "stderr": 4.444,
                "cost_per_test": 0.189466,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 262000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "deepseek/deepseek-v4-pro-0813": {
                "accuracy": 28.889,
                "latency": 2558.376,
                "stderr": 2.94,
                "cost_per_test": 0.287856,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 25.556,
                "latency": 2247.207,
                "stderr": 4.843,
                "cost_per_test": 0.214553,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 25.556,
                "latency": 2765.995,
                "stderr": 2.222,
                "cost_per_test": 0.410732,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "nvidia/nemotron-3-ultra-550b-a55b": {
                "accuracy": 24.444,
                "latency": 1732.877,
                "stderr": 4.006,
                "cost_per_test": null,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Nvidia",
                "harness": null
            },
            "ant/ling-3.0-flash-2607": {
                "accuracy": 23.333,
                "latency": 1761.684,
                "stderr": 1.925,
                "cost_per_test": 0.094613,
                "temperature": 0.6,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "alibaba/qwen3.7-plus": {
                "accuracy": 23.333,
                "latency": 1943.118,
                "stderr": 3.849,
                "cost_per_test": 0.284352,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "thinkingmachines/inkling": {
                "accuracy": 23.333,
                "latency": 2288.936,
                "stderr": 1.925,
                "cost_per_test": 0.947521,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 22.222,
                "latency": 1926.817,
                "stderr": 1.111,
                "cost_per_test": 1.079338,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 18.889,
                "latency": 2410.826,
                "stderr": 5.879,
                "cost_per_test": 0.195835,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 18.889,
                "latency": 3118.2,
                "stderr": 4.006,
                "cost_per_test": 0.254669,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 16.667,
                "latency": 906.987,
                "stderr": 1.925,
                "cost_per_test": 1.208509,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 15.556,
                "latency": 825.749,
                "stderr": 1.111,
                "cost_per_test": 0.251852,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "mistralai/mistral-medium-3.5": {
                "accuracy": 15.556,
                "latency": 2934.535,
                "stderr": 2.222,
                "cost_per_test": 3.533252,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "grok/grok-4.3": {
                "accuracy": 14.444,
                "latency": 1396.45,
                "stderr": 2.222,
                "cost_per_test": 2.913868,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "poolside/laguna-m.1": {
                "accuracy": 13.333,
                "latency": 3588.504,
                "stderr": 0.0,
                "cost_per_test": 0.891706,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "inception/mercury-2.5": {
                "accuracy": 12.222,
                "latency": 2043.805,
                "stderr": 2.222,
                "cost_per_test": 0.637794,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Inception",
                "harness": null
            },
            "poolside/laguna-xs.2": {
                "accuracy": 11.111,
                "latency": 3615.114,
                "stderr": 2.94,
                "cost_per_test": 0.752687,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "cohere/command-a-plus-05-2026": {
                "accuracy": 3.333,
                "latency": 0.0,
                "stderr": 0.0,
                "cost_per_test": null,
                "temperature": 0.9,
                "top_p": 0.95,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "fireworks/nemotron-lightning-3p5-30b-a3b": {
                "accuracy": 1.111,
                "latency": 3050.703,
                "stderr": 1.111,
                "cost_per_test": 0.042411,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "anthropic/claude-sonnet-5-5": {
                "accuracy": 76.667,
                "latency": 1381.955,
                "stderr": 5.092,
                "cost_per_test": 0.776061,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            }
        }
    }
}