{
    "metadata": {
        "benchmark": "Vals Multimodal Index",
        "slug": "vals_multimodal_index_v1",
        "description": "Benchmark consisting of a weighted performance across finance, coding, and education tasks. Showing the potential impact that LLM's can have on the economy.",
        "benchmark_id": "vals_multimodal_index_v1",
        "family": "vals_multimodal_index",
        "version": "1",
        "updated": "2026-05-04",
        "dataset_type": "private",
        "industry": "index",
        "tasks": {
            "overall": "Overall",
            "finance_agent": "Finance Agent v1.1",
            "corp_fin_v2": "CorpFin v2",
            "swebench": "SWE-bench Verified",
            "terminal_bench_2": "Terminal-Bench 2.0",
            "vibe_code_bench": "Vibe Code Bench",
            "sage": "SAGE",
            "mortgage_tax": "Mortgage Tax"
        },
        "models": [
            "alibaba/qwen3.5-plus-thinking",
            "alibaba/qwen3.6-27b",
            "alibaba/qwen3.6-plus",
            "anthropic/claude-haiku-4-5-20251001-thinking",
            "anthropic/claude-opus-4-5-20251101-thinking",
            "anthropic/claude-opus-4-6-thinking",
            "anthropic/claude-opus-4-7",
            "anthropic/claude-sonnet-4-5-20250929-thinking",
            "anthropic/claude-sonnet-4-6",
            "google/gemini-2.5-pro",
            "google/gemini-3-flash-preview",
            "google/gemini-3-pro-preview",
            "google/gemini-3.1-flash-lite-preview",
            "google/gemini-3.1-pro-preview",
            "grok/grok-4-1-fast-reasoning",
            "grok/grok-4-fast-reasoning",
            "grok/grok-4.20-0309-reasoning",
            "grok/grok-4.3",
            "kimi/kimi-k2.5-thinking",
            "kimi/kimi-k2.6",
            "openai/gpt-5-2025-08-07",
            "openai/gpt-5-mini-2025-08-07",
            "openai/gpt-5.1-2025-11-13",
            "openai/gpt-5.2-2025-12-11",
            "openai/gpt-5.4-2026-03-05",
            "openai/gpt-5.4-mini-2026-03-17",
            "openai/gpt-5.4-nano-2026-03-17",
            "openai/gpt-5.5"
        ],
        "partners": [],
        "showBadge": true,
        "visible": true,
        "use_cost_per_test": true,
        "runner": "platform",
        "mode": "agentic",
        "archived": true,
        "total_models": 28
    },
    "tasks": {
        "overall": {
            "anthropic/claude-opus-4-7": {
                "accuracy": 71.104,
                "latency": 534.875,
                "stderr": 1.511,
                "cost_per_test": 4.003889,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic"
            },
            "openai/gpt-5.5": {
                "accuracy": 69.84,
                "latency": 809.584,
                "stderr": 1.73,
                "cost_per_test": 3.460611,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 65.172,
                "latency": 1088.426,
                "stderr": 1.693,
                "cost_per_test": 2.020033,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 65.16,
                "latency": 459.23,
                "stderr": 1.7,
                "cost_per_test": 1.84156,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic"
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 64.001,
                "latency": 483.14,
                "stderr": 1.704,
                "cost_per_test": 1.433586,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic"
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 63.749,
                "latency": 1044.256,
                "stderr": 1.77,
                "cost_per_test": 3.027642,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "kimi/kimi-k2.6": {
                "accuracy": 59.103,
                "latency": 738.604,
                "stderr": 2.003,
                "cost_per_test": 0.4385,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI"
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 58.618,
                "latency": 354.486,
                "stderr": 1.718,
                "cost_per_test": 0.91081,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 55.927,
                "latency": 841.83,
                "stderr": 2.068,
                "cost_per_test": 0.48596,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 55.222,
                "latency": 515.342,
                "stderr": 1.484,
                "cost_per_test": 5.362427,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic"
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 54.331,
                "latency": 359.336,
                "stderr": 1.767,
                "cost_per_test": 0.260669,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 53.982,
                "latency": 609.8,
                "stderr": 1.689,
                "cost_per_test": 0.962961,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba"
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 53.078,
                "latency": 1648.925,
                "stderr": 1.427,
                "cost_per_test": 1.219845,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 52.697,
                "latency": 573.75,
                "stderr": 1.757,
                "cost_per_test": 0.625996,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 52.01,
                "latency": 612.693,
                "stderr": 1.498,
                "cost_per_test": 0.210836,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI"
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 51.983,
                "latency": 654.128,
                "stderr": 1.604,
                "cost_per_test": 1.412854,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 50.805,
                "latency": 908.182,
                "stderr": 1.976,
                "cost_per_test": 0.240585,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "alibaba/qwen3.6-27b": {
                "accuracy": 50.18,
                "latency": 1931.236,
                "stderr": 1.322,
                "cost_per_test": 0.929022,
                "temperature": 1,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba"
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 49.933,
                "latency": 631.208,
                "stderr": 1.536,
                "cost_per_test": 0.420847,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "alibaba/qwen3.5-plus-thinking": {
                "accuracy": 48.791,
                "latency": 878.599,
                "stderr": 1.47,
                "cost_per_test": 0.768932,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba"
            },
            "grok/grok-4.3": {
                "accuracy": 47.041,
                "latency": 572.468,
                "stderr": 1.586,
                "cost_per_test": 0.456985,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 46.576,
                "latency": 274.336,
                "stderr": 1.407,
                "cost_per_test": 0.368003,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 46.189,
                "latency": 390.533,
                "stderr": 1.493,
                "cost_per_test": 0.085512,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 43.858,
                "latency": 217.914,
                "stderr": 1.229,
                "cost_per_test": 0.112584,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 42.966,
                "latency": 110.255,
                "stderr": 1.275,
                "cost_per_test": 0.297821,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "google/gemini-2.5-pro": {
                "accuracy": 42.408,
                "latency": 724.818,
                "stderr": 1.234,
                "cost_per_test": 0.36268,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "grok/grok-4-fast-reasoning": {
                "accuracy": 38.367,
                "latency": 303.879,
                "stderr": 1.226,
                "cost_per_test": 0.046807,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 38.177,
                "latency": 199.494,
                "stderr": 1.21,
                "cost_per_test": 0.053562,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            }
        },
        "finance_agent": {
            "anthropic/claude-opus-4-7": {
                "accuracy": 72.222,
                "latency": 273.714,
                "stderr": 4.748,
                "cost_per_test": 1.33672,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic"
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 68.889,
                "latency": 347.536,
                "stderr": 4.907,
                "cost_per_test": 1.334413,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic"
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 64.444,
                "latency": 312.428,
                "stderr": 5.074,
                "cost_per_test": 1.119515,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic"
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 63.333,
                "latency": 593.047,
                "stderr": 5.108,
                "cost_per_test": 0.99647,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "openai/gpt-5.5": {
                "accuracy": 63.333,
                "latency": 899.814,
                "stderr": 5.108,
                "cost_per_test": 3.732358,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 62.222,
                "latency": 210.976,
                "stderr": 5.139,
                "cost_per_test": 1.115022,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 60,
                "latency": 190.087,
                "stderr": 5.193,
                "cost_per_test": 1.531755,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic"
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 60,
                "latency": 742.706,
                "stderr": 5.193,
                "cost_per_test": 1.555334,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 60,
                "latency": 927.737,
                "stderr": 5.193,
                "cost_per_test": 0.517677,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 58.889,
                "latency": 150.386,
                "stderr": 5.216,
                "cost_per_test": 0.610614,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 58.889,
                "latency": 284.479,
                "stderr": 5.216,
                "cost_per_test": 0.976233,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 58.889,
                "latency": 307.156,
                "stderr": 5.216,
                "cost_per_test": 0.152637,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba"
            },
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 58.889,
                "latency": 569.69,
                "stderr": 5.216,
                "cost_per_test": 0.483019,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 57.778,
                "latency": 265.478,
                "stderr": 5.235,
                "cost_per_test": 0.174563,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI"
            },
            "alibaba/qwen3.5-plus-thinking": {
                "accuracy": 57.778,
                "latency": 334.677,
                "stderr": 5.235,
                "cost_per_test": 0.256637,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba"
            },
            "kimi/kimi-k2.6": {
                "accuracy": 57.778,
                "latency": 532.653,
                "stderr": 5.235,
                "cost_per_test": 0.489502,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI"
            },
            "grok/grok-4.3": {
                "accuracy": 57.778,
                "latency": 772.276,
                "stderr": 5.235,
                "cost_per_test": 0.427426,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 56.667,
                "latency": 215.762,
                "stderr": 5.253,
                "cost_per_test": 0.049297,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 55.556,
                "latency": 98.738,
                "stderr": 5.267,
                "cost_per_test": 0.064312,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 55.556,
                "latency": 855.084,
                "stderr": 5.267,
                "cost_per_test": 0.343008,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "alibaba/qwen3.6-27b": {
                "accuracy": 55,
                "latency": 584.833,
                "stderr": 3.727,
                "cost_per_test": 0.39447,
                "temperature": 1,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba"
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 54.444,
                "latency": 113.952,
                "stderr": 5.279,
                "cost_per_test": 0.364612,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 53.333,
                "latency": 1047.445,
                "stderr": 5.288,
                "cost_per_test": 0.507004,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 52.222,
                "latency": 70.486,
                "stderr": 5.295,
                "cost_per_test": 0.353635,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 52.222,
                "latency": 640.481,
                "stderr": 5.295,
                "cost_per_test": 0.150327,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "grok/grok-4-fast-reasoning": {
                "accuracy": 48.889,
                "latency": 60.939,
                "stderr": 5.299,
                "cost_per_test": 0.049119,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 48.889,
                "latency": 122.395,
                "stderr": 5.299,
                "cost_per_test": 0.072996,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "google/gemini-2.5-pro": {
                "accuracy": 43.333,
                "latency": 1834.194,
                "stderr": 5.253,
                "cost_per_test": 0.324011,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            }
        },
        "corp_fin_v2": {
            "grok/grok-4-fast-reasoning": {
                "accuracy": 71.096,
                "latency": 10.442,
                "stderr": 1.548,
                "cost_per_test": 0.043356,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 70.163,
                "latency": 69.331,
                "stderr": 1.562,
                "cost_per_test": 0.043611,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI"
            },
            "grok/grok-4.3": {
                "accuracy": 70.047,
                "latency": 24.725,
                "stderr": 1.564,
                "cost_per_test": 0.081293,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "openai/gpt-5.5": {
                "accuracy": 69.58,
                "latency": 26.764,
                "stderr": 1.571,
                "cost_per_test": 0.43166,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 68.998,
                "latency": 22.376,
                "stderr": 1.579,
                "cost_per_test": 0.478497,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic"
            },
            "kimi/kimi-k2.6": {
                "accuracy": 68.182,
                "latency": 96.974,
                "stderr": 1.59,
                "cost_per_test": 0.042064,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 48000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI"
            },
            "alibaba/qwen3.5-plus-thinking": {
                "accuracy": 67.832,
                "latency": 40.886,
                "stderr": 1.595,
                "cost_per_test": 0.034482,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba"
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 67.716,
                "latency": 14.833,
                "stderr": 1.596,
                "cost_per_test": 0.039017,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 67.599,
                "latency": 18.561,
                "stderr": 1.598,
                "cost_per_test": 0.012735,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 66.783,
                "latency": 122.651,
                "stderr": 1.608,
                "cost_per_test": 0.235266,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 66.667,
                "latency": 20.037,
                "stderr": 1.609,
                "cost_per_test": 0.287644,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic"
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 66.2,
                "latency": 20.361,
                "stderr": 1.615,
                "cost_per_test": 0.475505,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic"
            },
            "anthropic/claude-opus-4-7": {
                "accuracy": 66.084,
                "latency": 18.221,
                "stderr": 1.616,
                "cost_per_test": 0.816017,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic"
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 65.851,
                "latency": 9.418,
                "stderr": 1.619,
                "cost_per_test": 0.130871,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 65.734,
                "latency": 28.137,
                "stderr": 1.62,
                "cost_per_test": 0.145587,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 65.734,
                "latency": 31.057,
                "stderr": 1.62,
                "cost_per_test": 0.109551,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 65.035,
                "latency": 32.888,
                "stderr": 1.628,
                "cost_per_test": 0.090475,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 64.569,
                "latency": 23.778,
                "stderr": 1.633,
                "cost_per_test": 0.154515,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 64.219,
                "latency": 21.997,
                "stderr": 1.636,
                "cost_per_test": 0.286695,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 63.17,
                "latency": 15.049,
                "stderr": 1.647,
                "cost_per_test": 0.096327,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "google/gemini-2.5-pro": {
                "accuracy": 62.587,
                "latency": 28.746,
                "stderr": 1.652,
                "cost_per_test": 0.081072,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "alibaba/qwen3.6-27b": {
                "accuracy": 62.354,
                "latency": 51.777,
                "stderr": 1.654,
                "cost_per_test": 0.046263,
                "temperature": 1,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba"
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 62.354,
                "latency": 158.5,
                "stderr": 1.654,
                "cost_per_test": 0.079354,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 62.238,
                "latency": 35.919,
                "stderr": 1.655,
                "cost_per_test": 0.038254,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba"
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 61.189,
                "latency": 109.941,
                "stderr": 1.664,
                "cost_per_test": 0.019795,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 61.072,
                "latency": 7.363,
                "stderr": 1.665,
                "cost_per_test": 0.01612,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 60.839,
                "latency": 4.708,
                "stderr": 1.666,
                "cost_per_test": 0.022927,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 60.256,
                "latency": 45.279,
                "stderr": 1.671,
                "cost_per_test": 0.068348,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            }
        },
        "swebench": {
            "openai/gpt-5.5": {
                "accuracy": 82.6,
                "latency": 426.425,
                "stderr": 1.695,
                "cost_per_test": 1.361765,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": null
            },
            "anthropic/claude-opus-4-7": {
                "accuracy": 82.353,
                "latency": 508.652,
                "stderr": 3.775,
                "cost_per_test": 2.999546,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": null
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 77.451,
                "latency": 633.682,
                "stderr": 4.138,
                "cost_per_test": 1.706873,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": null
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 75.49,
                "latency": 353.853,
                "stderr": 4.259,
                "cost_per_test": 0.94669,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 74.51,
                "latency": 396.661,
                "stderr": 4.315,
                "cost_per_test": 0.322908,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 72.549,
                "latency": 329.202,
                "stderr": 4.419,
                "cost_per_test": 0.820116,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": null
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 72.549,
                "latency": 407.124,
                "stderr": 4.419,
                "cost_per_test": 1.459323,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": null
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 72.549,
                "latency": 430.643,
                "stderr": 4.419,
                "cost_per_test": 0.847172,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": null
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 71.569,
                "latency": 336.507,
                "stderr": 4.466,
                "cost_per_test": 1.202144,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": null
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 71.569,
                "latency": 722.109,
                "stderr": 4.466,
                "cost_per_test": 1.706049,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": null
            },
            "grok/grok-4.3": {
                "accuracy": 70.588,
                "latency": 126.579,
                "stderr": 4.512,
                "cost_per_test": 0.23972,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 70.588,
                "latency": 487.435,
                "stderr": 4.512,
                "cost_per_test": 0.901327,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": null
            },
            "alibaba/qwen3.6-27b": {
                "accuracy": 70.588,
                "latency": 910.234,
                "stderr": 4.512,
                "cost_per_test": 1.041977,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": null
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 68.627,
                "latency": 327.172,
                "stderr": 4.594,
                "cost_per_test": 0.306186,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 67.647,
                "latency": 281.133,
                "stderr": 4.632,
                "cost_per_test": 0.107316,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": null
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 67.647,
                "latency": 355.599,
                "stderr": 4.632,
                "cost_per_test": 0.589598,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": null
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 67.647,
                "latency": 388.838,
                "stderr": 4.632,
                "cost_per_test": 0.890462,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 66.667,
                "latency": 151.343,
                "stderr": 4.668,
                "cost_per_test": 0.586871,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": null
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 66.667,
                "latency": 374.288,
                "stderr": 4.668,
                "cost_per_test": 0.194738,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": null
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 65.686,
                "latency": 252.837,
                "stderr": 4.701,
                "cost_per_test": 0.296208,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": null
            },
            "alibaba/qwen3.5-plus-thinking": {
                "accuracy": 65.686,
                "latency": 507.258,
                "stderr": 4.701,
                "cost_per_test": 0.917096,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": null
            },
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 65.686,
                "latency": 546.856,
                "stderr": 4.701,
                "cost_per_test": 0.794681,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 64.706,
                "latency": 264.186,
                "stderr": 4.732,
                "cost_per_test": 0.407017,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 57.843,
                "latency": 136.077,
                "stderr": 4.889,
                "cost_per_test": 0.106331,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": null
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 55.882,
                "latency": 195.055,
                "stderr": 4.916,
                "cost_per_test": 0.053604,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": null
            },
            "google/gemini-2.5-pro": {
                "accuracy": 49.02,
                "latency": 240.174,
                "stderr": 4.95,
                "cost_per_test": 0.424246,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": null
            },
            "grok/grok-4-fast-reasoning": {
                "accuracy": 42.157,
                "latency": 129.441,
                "stderr": 4.889,
                "cost_per_test": 0.051519,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": null
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 36.275,
                "latency": 147.938,
                "stderr": 4.761,
                "cost_per_test": 0.029005,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": null
            }
        },
        "terminal_bench_2": {
            "openai/gpt-5.5": {
                "accuracy": 73.202,
                "latency": 2298.224,
                "stderr": 3.978,
                "cost_per_test": 1.819086,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "anthropic/claude-opus-4-7": {
                "accuracy": 68.539,
                "latency": 660.89,
                "stderr": 4.95,
                "cost_per_test": 0.975907,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic"
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 67.416,
                "latency": 542.887,
                "stderr": 4.996,
                "cost_per_test": 0.516084,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 59.551,
                "latency": 615.5,
                "stderr": 5.232,
                "cost_per_test": 0.518966,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": false,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic"
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 58.427,
                "latency": 898.494,
                "stderr": 5.254,
                "cost_per_test": 1.140327,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic"
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 58.427,
                "latency": 1027.517,
                "stderr": 5.254,
                "cost_per_test": 0.511959,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "kimi/kimi-k2.6": {
                "accuracy": 57.303,
                "latency": 795.458,
                "stderr": 5.273,
                "cost_per_test": 0.192786,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI"
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 55.056,
                "latency": 416.249,
                "stderr": 5.303,
                "cost_per_test": 0.455782,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 53.933,
                "latency": 661.913,
                "stderr": 5.314,
                "cost_per_test": 1.122133,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic"
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 51.685,
                "latency": 456.071,
                "stderr": 5.327,
                "cost_per_test": 0.168702,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 51.685,
                "latency": 742.684,
                "stderr": 5.327,
                "cost_per_test": 0.469958,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 44.944,
                "latency": 840.666,
                "stderr": 5.303,
                "cost_per_test": 0.14597,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba"
            },
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 44.944,
                "latency": 886.169,
                "stderr": 5.303,
                "cost_per_test": 0.407628,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "alibaba/qwen3.6-27b": {
                "accuracy": 44.944,
                "latency": 2007.631,
                "stderr": 5.303,
                "cost_per_test": 0.298145,
                "temperature": 0.6,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba"
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 44.944,
                "latency": 2241.556,
                "stderr": 4.492,
                "cost_per_test": 0.914107,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "grok/grok-4.3": {
                "accuracy": 43.446,
                "latency": 1954.824,
                "stderr": 4.406,
                "cost_per_test": 1.115571,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 41.573,
                "latency": 736.918,
                "stderr": 5.254,
                "cost_per_test": 0.751119,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "alibaba/qwen3.5-plus-thinking": {
                "accuracy": 41.573,
                "latency": 1877.151,
                "stderr": 5.254,
                "cost_per_test": 0.321994,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba"
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 40.449,
                "latency": 161.408,
                "stderr": 5.232,
                "cost_per_test": 0.154109,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 40.449,
                "latency": 756.405,
                "stderr": 5.232,
                "cost_per_test": 0.139997,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI"
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 39.888,
                "latency": 2527.932,
                "stderr": 4.454,
                "cost_per_test": 0.224388,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 38.202,
                "latency": 655.044,
                "stderr": 5.18,
                "cost_per_test": 0.327722,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 37.079,
                "latency": 986.102,
                "stderr": 5.149,
                "cost_per_test": 0.477064,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "google/gemini-2.5-pro": {
                "accuracy": 30.337,
                "latency": 796.526,
                "stderr": 4.901,
                "cost_per_test": 0.472003,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "grok/grok-4-fast-reasoning": {
                "accuracy": 29.213,
                "latency": 431.268,
                "stderr": 4.848,
                "cost_per_test": 0.03431,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 26.966,
                "latency": 1028.621,
                "stderr": 4.731,
                "cost_per_test": 0.116731,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 24.719,
                "latency": 479.157,
                "stderr": 4.599,
                "cost_per_test": 0.055526,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 24.719,
                "latency": 667.879,
                "stderr": 4.599,
                "cost_per_test": 0.065496,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            }
        },
        "sage": {
            "anthropic/claude-opus-4-7": {
                "accuracy": 56.103,
                "latency": 124.647,
                "stderr": 3.368,
                "cost_per_test": 0.407416,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic"
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 52.092,
                "latency": 87.217,
                "stderr": 3.403,
                "cost_per_test": 0.153779,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic"
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 51.849,
                "latency": 42.697,
                "stderr": 3.404,
                "cost_per_test": 0.018229,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 51.575,
                "latency": 160.968,
                "stderr": 3.339,
                "cost_per_test": 0.353448,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic"
            },
            "openai/gpt-5.5": {
                "accuracy": 51.532,
                "latency": 76.05,
                "stderr": 3.954,
                "cost_per_test": 0.14609,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 50.813,
                "latency": 141.066,
                "stderr": 3.403,
                "cost_per_test": 0.075919,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "kimi/kimi-k2.6": {
                "accuracy": 50.224,
                "latency": 270.956,
                "stderr": 3.429,
                "cost_per_test": 0.068198,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI"
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 49.865,
                "latency": 173.179,
                "stderr": 3.363,
                "cost_per_test": 0.028736,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI"
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 49.54,
                "latency": 16.257,
                "stderr": 3.484,
                "cost_per_test": 0.0075,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 49.27,
                "latency": 178.887,
                "stderr": 3.354,
                "cost_per_test": 0.115807,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 48.677,
                "latency": 61.174,
                "stderr": 3.293,
                "cost_per_test": 0.068908,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 47.615,
                "latency": 97.275,
                "stderr": 3.376,
                "cost_per_test": 0.012041,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 46.582,
                "latency": 163.365,
                "stderr": 3.43,
                "cost_per_test": 0.234647,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic"
            },
            "alibaba/qwen3.6-27b": {
                "accuracy": 45.577,
                "latency": 159.25,
                "stderr": 3.427,
                "cost_per_test": 0.050269,
                "temperature": 1,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba"
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 44.86,
                "latency": 229.267,
                "stderr": 3.433,
                "cost_per_test": 0.041429,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba"
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 43.68,
                "latency": 58.593,
                "stderr": 3.349,
                "cost_per_test": 0.032224,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 43.312,
                "latency": 152.404,
                "stderr": 3.115,
                "cost_per_test": 0.100559,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 43.235,
                "latency": 98.207,
                "stderr": 3.16,
                "cost_per_test": 0.004838,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 42.988,
                "latency": 36.524,
                "stderr": 3.338,
                "cost_per_test": 0.005183,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "google/gemini-2.5-pro": {
                "accuracy": 41.916,
                "latency": 52.934,
                "stderr": 3.411,
                "cost_per_test": 0.008806,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 38.242,
                "latency": 66.358,
                "stderr": 3.425,
                "cost_per_test": 0.063231,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 38.081,
                "latency": 34.334,
                "stderr": 3.087,
                "cost_per_test": 0.007088,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 36.065,
                "latency": 210.409,
                "stderr": 3.209,
                "cost_per_test": 0.137598,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 31.822,
                "latency": 66.445,
                "stderr": 3.134,
                "cost_per_test": 0.05313,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 31.331,
                "latency": 77.053,
                "stderr": 3.02,
                "cost_per_test": 0.001473,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "alibaba/qwen3.5-plus-thinking": {
                "accuracy": 30.402,
                "latency": 312.529,
                "stderr": 3.147,
                "cost_per_test": 0.03774,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba"
            },
            "grok/grok-4-fast-reasoning": {
                "accuracy": 29.761,
                "latency": 31.734,
                "stderr": 3.043,
                "cost_per_test": 0.011583,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "grok/grok-4.3": {
                "accuracy": 19.736,
                "latency": 96.2,
                "stderr": 2.647,
                "cost_per_test": 0.037978,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            }
        },
        "mortgage_tax": {
            "anthropic/claude-opus-4-7": {
                "accuracy": 70.27,
                "latency": 16.101,
                "stderr": 0.896,
                "cost_per_test": 0.084417,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic"
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 69.396,
                "latency": 23.616,
                "stderr": 0.908,
                "cost_per_test": 0.023542,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 69.078,
                "latency": 25.602,
                "stderr": 0.911,
                "cost_per_test": 0.041198,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "google/gemini-2.5-pro": {
                "accuracy": 68.918,
                "latency": 24.008,
                "stderr": 0.907,
                "cost_per_test": 0.004794,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "openai/gpt-5.5": {
                "accuracy": 68.76,
                "latency": 28.243,
                "stderr": 0.912,
                "cost_per_test": 0.071407,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 68.72,
                "latency": 12.838,
                "stderr": 0.914,
                "cost_per_test": 0.007477,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 68.522,
                "latency": 27.161,
                "stderr": 0.911,
                "cost_per_test": 0.060749,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic"
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 68.323,
                "latency": 45.002,
                "stderr": 0.915,
                "cost_per_test": 0.103445,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "alibaba/qwen3.6-27b": {
                "accuracy": 68.283,
                "latency": 42.074,
                "stderr": 0.916,
                "cost_per_test": 0.016906,
                "temperature": 1,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba"
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 68.044,
                "latency": 5.368,
                "stderr": 0.912,
                "cost_per_test": 0.00284,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 67.965,
                "latency": 78.18,
                "stderr": 0.908,
                "cost_per_test": 0.013476,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba"
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 67.726,
                "latency": 29.74,
                "stderr": 0.921,
                "cost_per_test": 0.04467,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic"
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 67.686,
                "latency": 28.166,
                "stderr": 0.917,
                "cost_per_test": 0.178388,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic"
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 67.13,
                "latency": 70.666,
                "stderr": 0.918,
                "cost_per_test": 0.050478,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 66.892,
                "latency": 24.864,
                "stderr": 0.923,
                "cost_per_test": 0.00436,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 66.534,
                "latency": 79.787,
                "stderr": 0.925,
                "cost_per_test": 0.014918,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI"
            },
            "kimi/kimi-k2.6": {
                "accuracy": 65.818,
                "latency": 109.76,
                "stderr": 0.93,
                "cost_per_test": 0.02869,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI"
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 65.454,
                "latency": 62.828,
                "stderr": 0.879,
                "cost_per_test": 0.028906,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 63.99,
                "latency": 47.622,
                "stderr": 0.955,
                "cost_per_test": 0.052096,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 63.514,
                "latency": 126.454,
                "stderr": 0.906,
                "cost_per_test": 0.044769,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 62.162,
                "latency": 29.792,
                "stderr": 0.962,
                "cost_per_test": 0.020742,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 61.368,
                "latency": 46.254,
                "stderr": 0.931,
                "cost_per_test": 0.026408,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "alibaba/qwen3.5-plus-thinking": {
                "accuracy": 60.772,
                "latency": 62.075,
                "stderr": 0.972,
                "cost_per_test": 0.010812,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba"
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 59.102,
                "latency": 14.526,
                "stderr": 0.964,
                "cost_per_test": 0.002585,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "grok/grok-4.3": {
                "accuracy": 48.252,
                "latency": 443.266,
                "stderr": 0.973,
                "cost_per_test": 0.012184,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 45.35,
                "latency": 11.124,
                "stderr": 0.988,
                "cost_per_test": 0.024144,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 42.607,
                "latency": 47.45,
                "stderr": 0.952,
                "cost_per_test": 0.002438,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "grok/grok-4-fast-reasoning": {
                "accuracy": 42.09,
                "latency": 14.475,
                "stderr": 0.95,
                "cost_per_test": 0.004554,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            }
        },
        "vibe_code_bench": {
            "anthropic/claude-opus-4-7": {
                "accuracy": 77.701,
                "latency": 2141.902,
                "stderr": 5.336,
                "cost_per_test": 21.407199,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic"
            },
            "openai/gpt-5.5": {
                "accuracy": 77.099,
                "latency": 1911.567,
                "stderr": 7.028,
                "cost_per_test": 16.661911,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 73.143,
                "latency": 5174.848,
                "stderr": 6.271,
                "cost_per_test": 10.686976,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 67.361,
                "latency": 4971.341,
                "stderr": 6.821,
                "cost_per_test": 17.745182,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 64.478,
                "latency": 1386.056,
                "stderr": 6.364,
                "cost_per_test": 8.279062,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic"
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 56.223,
                "latency": 1572.118,
                "stderr": 6.546,
                "cost_per_test": 5.907892,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic"
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 42.523,
                "latency": 2055.117,
                "stderr": 8.901,
                "cost_per_test": 1.191299,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "kimi/kimi-k2.6": {
                "accuracy": 42.323,
                "latency": 2967.768,
                "stderr": 8.392,
                "cost_per_test": 1.925351,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI"
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 34.214,
                "latency": 1211.906,
                "stderr": 6.444,
                "cost_per_test": 3.825202,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 30.032,
                "latency": 3276.224,
                "stderr": 8.282,
                "cost_per_test": 1.277298,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 28.253,
                "latency": 2289.977,
                "stderr": 6.126,
                "cost_per_test": 5.447636,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba"
            },
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 28.231,
                "latency": 1836.188,
                "stderr": 6.633,
                "cost_per_test": 2.574926,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 23.337,
                "latency": 2962.133,
                "stderr": 5.528,
                "cost_per_test": 6.656984,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 22.156,
                "latency": 2283.142,
                "stderr": 4.45,
                "cost_per_test": 32.873283,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic"
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 21.954,
                "latency": 806.66,
                "stderr": 6.673,
                "cost_per_test": 0.942067,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 21.314,
                "latency": 1852.149,
                "stderr": 4.816,
                "cost_per_test": 1.525167,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "grok/grok-4.3": {
                "accuracy": 15.475,
                "latency": 589.407,
                "stderr": 5.576,
                "cost_per_test": 1.28472,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 14.769,
                "latency": 2570.381,
                "stderr": 4.525,
                "cost_per_test": 0.879292,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI"
            },
            "alibaba/qwen3.5-plus-thinking": {
                "accuracy": 13.731,
                "latency": 3015.618,
                "stderr": 4.258,
                "cost_per_test": 3.803761,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba"
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 12.978,
                "latency": 10398.545,
                "stderr": 3.879,
                "cost_per_test": 6.417592,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 12.576,
                "latency": 698.246,
                "stderr": 4.46,
                "cost_per_test": 0.248587,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "alibaba/qwen3.6-27b": {
                "accuracy": 11.142,
                "latency": 9762.856,
                "stderr": 4.329,
                "cost_per_test": 4.655127,
                "temperature": 1,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba"
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 9.882,
                "latency": 775.882,
                "stderr": 3.549,
                "cost_per_test": 1.306474,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 1.515,
                "latency": 301.651,
                "stderr": 1.515,
                "cost_per_test": 0.771885,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 0,
                "latency": 527.562,
                "stderr": 0,
                "cost_per_test": 0.209444,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 0,
                "latency": 572.711,
                "stderr": 0,
                "cost_per_test": 0.51,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "grok/grok-4-fast-reasoning": {
                "accuracy": 0,
                "latency": 1448.855,
                "stderr": 0,
                "cost_per_test": 0.133206,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "google/gemini-2.5-pro": {
                "accuracy": 0,
                "latency": 2097.144,
                "stderr": 0,
                "cost_per_test": 1.223831,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            }
        }
    }
}
