{
    "metadata": {
        "benchmark": "MedQA",
        "slug": "medqa",
        "description": "Evaluating language model bias in medical questions.",
        "benchmark_id": "medqa",
        "family": "medqa",
        "version": "1",
        "updated": "2026-04-16",
        "dataset_type": "public",
        "industry": "healthcare",
        "tasks": {
            "overall": "Overall",
            "unbiased": "Unbiased",
            "hispanic": "Hispanic",
            "black": "Black",
            "asian": "Asian",
            "white": "White",
            "indigenous": "Indigenous"
        },
        "models": [
            "ai21labs/jamba-1.5-large",
            "ai21labs/jamba-1.5-mini",
            "ai21labs/jamba-large-1.6",
            "ai21labs/jamba-mini-1.6",
            "alibaba/qwen3-max",
            "alibaba/qwen3-max-preview",
            "alibaba/qwen3.5-plus-thinking",
            "anthropic/claude-3-5-sonnet-20241022",
            "anthropic/claude-3-7-sonnet-20250219-thinking",
            "anthropic/claude-haiku-4-5-20251001-thinking",
            "anthropic/claude-opus-4-1-20250805",
            "anthropic/claude-opus-4-1-20250805-thinking",
            "anthropic/claude-opus-4-20250514",
            "anthropic/claude-opus-4-5-20251101",
            "anthropic/claude-opus-4-5-20251101-thinking",
            "anthropic/claude-opus-4-6-thinking",
            "anthropic/claude-sonnet-4-20250514",
            "anthropic/claude-sonnet-4-20250514-thinking",
            "anthropic/claude-sonnet-4-5-20250929-thinking",
            "anthropic/claude-sonnet-4-6",
            "cohere/command-a-03-2025",
            "cohere/command-r-plus",
            "fireworks/deepseek-r1",
            "fireworks/deepseek-v3",
            "fireworks/deepseek-v3-0324",
            "fireworks/deepseek-v3p2",
            "fireworks/deepseek-v3p2-thinking",
            "fireworks/gpt-oss-120b",
            "fireworks/gpt-oss-20b",
            "fireworks/llama4-maverick-instruct-basic",
            "fireworks/qwen3-235b-a22b",
            "google/gemini-1.5-pro-002",
            "google/gemini-2.0-flash-001",
            "google/gemini-2.5-flash-lite-preview-09-2025",
            "google/gemini-2.5-flash-lite-preview-09-2025-thinking",
            "google/gemini-2.5-flash-preview-04-17",
            "google/gemini-2.5-flash-preview-04-17-thinking",
            "google/gemini-2.5-flash-preview-09-2025",
            "google/gemini-2.5-flash-preview-09-2025-thinking",
            "google/gemini-2.5-pro-exp-03-25",
            "google/gemini-3-flash-preview",
            "google/gemini-3-pro-preview",
            "google/gemini-3.1-pro-preview",
            "grok/grok-2-1212",
            "grok/grok-3",
            "grok/grok-3-mini-fast-high-reasoning",
            "grok/grok-3-mini-fast-low-reasoning",
            "grok/grok-4-0709",
            "grok/grok-4-1-fast-non-reasoning",
            "grok/grok-4-1-fast-reasoning",
            "grok/grok-4-fast-non-reasoning",
            "grok/grok-4-fast-reasoning",
            "grok/grok-4.20-0309-reasoning",
            "kimi/kimi-k2-thinking",
            "kimi/kimi-k2.5-thinking",
            "minimax/MiniMax-M2.1",
            "minimax/MiniMax-M2.5",
            "mistralai/magistral-medium-2509",
            "mistralai/magistral-small-2509",
            "mistralai/mistral-large-2411",
            "mistralai/mistral-large-2512",
            "mistralai/mistral-medium-2505",
            "mistralai/mistral-small-2402",
            "mistralai/mistral-small-2503",
            "openai/gpt-3.5-turbo",
            "openai/gpt-4-turbo",
            "openai/gpt-4.1-2025-04-14",
            "openai/gpt-4.1-mini-2025-04-14",
            "openai/gpt-4.1-nano-2025-04-14",
            "openai/gpt-4o-2024-08-06",
            "openai/gpt-4o-mini-2024-07-18",
            "openai/gpt-5-2025-08-07",
            "openai/gpt-5-mini-2025-08-07",
            "openai/gpt-5-nano-2025-08-07",
            "openai/gpt-5.1-2025-11-13",
            "openai/gpt-5.2-2025-12-11",
            "openai/gpt-5.4-2026-03-05",
            "openai/o1-2024-12-17",
            "openai/o1-mini-2024-09-12",
            "openai/o1-preview-2024-09-12",
            "openai/o3-2025-04-16",
            "openai/o3-mini-2025-01-31",
            "openai/o4-mini-2025-04-16",
            "together/Qwen/Qwen2.5-72B-Instruct-Turbo",
            "together/meta-llama/Llama-4-Scout-17B-16E-Instruct",
            "together/meta-llama/Meta-Llama-3.1-405B-Instruct-Turbo",
            "together/meta-llama/Meta-Llama-3.1-70B-Instruct-Turbo",
            "together/meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo",
            "together/mistralai/Mixtral-8x22B-Instruct-v0.1",
            "together/mistralai/Mixtral-8x7B-v0.1",
            "together/moonshotai/Kimi-K2-Instruct",
            "zai/glm-4.5",
            "zai/glm-4.6",
            "zai/glm-4.7",
            "zai/glm-5-thinking"
        ],
        "partners": [
            {
                "name": "graphite",
                "url": "https://www.thegraphitegroup.com/"
            }
        ],
        "showBadge": false,
        "visible": true,
        "use_cost_per_test": false,
        "runner": "platform",
        "mode": "one-shot",
        "archived": false,
        "total_models": 95
    },
    "tasks": {
        "overall": {
            "openai/o1-2024-12-17": {
                "accuracy": 96.517,
                "latency": 11.152,
                "stderr": 0.33,
                "cost_per_test": 0.05282,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 96.383,
                "latency": 12.291,
                "stderr": 0.17,
                "cost_per_test": 0.009789,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 96.367,
                "latency": 38.67,
                "stderr": 0.171,
                "cost_per_test": 0.017689,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 96.317,
                "latency": 35.496,
                "stderr": 0.172,
                "cost_per_test": 0.020888,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 96.092,
                "latency": 22.784,
                "stderr": 0.177,
                "cost_per_test": 0.043618,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "openai/o3-2025-04-16": {
                "accuracy": 96.058,
                "latency": 8.507,
                "stderr": 0.178,
                "cost_per_test": 0.00542,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 96.058,
                "latency": 16.253,
                "stderr": 0.178,
                "cost_per_test": 0.002953,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 96.033,
                "latency": 21.603,
                "stderr": 0.178,
                "cost_per_test": 0.029109,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "openai/o4-mini-2025-04-16": {
                "accuracy": 96.017,
                "latency": 6.604,
                "stderr": 0.178,
                "cost_per_test": 0.002784,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 95.875,
                "latency": 31.625,
                "stderr": 0.182,
                "cost_per_test": 0.14813,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 95.808,
                "latency": 26.324,
                "stderr": 0.183,
                "cost_per_test": 0.012699,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 95.408,
                "latency": 30.085,
                "stderr": 0.191,
                "cost_per_test": 0.026693,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "alibaba/qwen3.5-plus-thinking": {
                "accuracy": 95.208,
                "latency": 52.361,
                "stderr": 0.195,
                "cost_per_test": 0.00829,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba"
            },
            "openai/o3-mini-2025-01-31": {
                "accuracy": 94.833,
                "latency": 8.387,
                "stderr": 0.397,
                "cost_per_test": 0.005327,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 94.708,
                "latency": 35.776,
                "stderr": 0.204,
                "cost_per_test": 0.031308,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 94.55,
                "latency": 12.225,
                "stderr": 0.207,
                "cost_per_test": 0.009106,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 94.367,
                "latency": 41.628,
                "stderr": 0.21,
                "cost_per_test": 0.006992,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI"
            },
            "zai/glm-5-thinking": {
                "accuracy": 94.267,
                "latency": 81.08,
                "stderr": 0.212,
                "cost_per_test": 0.008007,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI"
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 94.133,
                "latency": 19.483,
                "stderr": 0.214,
                "cost_per_test": 0.021634,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "fireworks/deepseek-v3p2-thinking": {
                "accuracy": 93.917,
                "latency": 53.69,
                "stderr": 0.218,
                "cost_per_test": 0.000772,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 20480,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "zai/glm-4.7": {
                "accuracy": 93.742,
                "latency": 66.008,
                "stderr": 0.221,
                "cost_per_test": 0.00487,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI"
            },
            "anthropic/claude-opus-4-1-20250805-thinking": {
                "accuracy": 93.592,
                "latency": 23.61,
                "stderr": 0.224,
                "cost_per_test": 0.088223,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "openai/gpt-5-nano-2025-08-07": {
                "accuracy": 93.258,
                "latency": 20.462,
                "stderr": 0.229,
                "cost_per_test": 0.001254,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "anthropic/claude-opus-4-5-20251101": {
                "accuracy": 93.158,
                "latency": 10.357,
                "stderr": 0.23,
                "cost_per_test": 0.043201,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "google/gemini-2.5-pro-exp-03-25": {
                "accuracy": 93.142,
                "latency": 10.704,
                "stderr": 0.231,
                "cost_per_test": 0.002896,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "openai/o1-preview-2024-09-12": {
                "accuracy": 93.008,
                "latency": 16.443,
                "stderr": 0.456,
                "cost_per_test": 0.108667,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "anthropic/claude-opus-4-20250514": {
                "accuracy": 92.867,
                "latency": 11.929,
                "stderr": 0.235,
                "cost_per_test": 0.037936,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "anthropic/claude-sonnet-4-20250514-thinking": {
                "accuracy": 92.708,
                "latency": 26.989,
                "stderr": 0.237,
                "cost_per_test": 0.025234,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "kimi/kimi-k2-thinking": {
                "accuracy": 92.592,
                "latency": 225.305,
                "stderr": 0.239,
                "cost_per_test": 0.006153,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI"
            },
            "anthropic/claude-opus-4-1-20250805": {
                "accuracy": 92.533,
                "latency": 13.971,
                "stderr": 0.24,
                "cost_per_test": 0.041528,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "minimax/MiniMax-M2.5": {
                "accuracy": 92.525,
                "latency": 33.54,
                "stderr": 0.24,
                "cost_per_test": 0.002928,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax"
            },
            "grok/grok-4-0709": {
                "accuracy": 92.492,
                "latency": 64.247,
                "stderr": 0.24,
                "cost_per_test": 0.025753,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "grok/grok-2-1212": {
                "accuracy": 92.317,
                "latency": 4.093,
                "stderr": 0.477,
                "cost_per_test": 0.003276,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 0,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "zai/glm-4.6": {
                "accuracy": 92.225,
                "latency": 28.734,
                "stderr": 0.244,
                "cost_per_test": 0.004252,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 122880,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI"
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 92.083,
                "latency": 17.636,
                "stderr": 0.246,
                "cost_per_test": 0.000571,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "grok/grok-4-fast-reasoning": {
                "accuracy": 92.067,
                "latency": 5.343,
                "stderr": 0.247,
                "cost_per_test": 0.000744,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 92.058,
                "latency": 35.119,
                "stderr": 0.247,
                "cost_per_test": 0.035415,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "google/gemini-2.5-flash-preview-09-2025": {
                "accuracy": 91.433,
                "latency": 8.534,
                "stderr": 0.255,
                "cost_per_test": 0.001272,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "fireworks/gpt-oss-120b": {
                "accuracy": 91.36,
                "latency": 12.383,
                "stderr": 0.24,
                "cost_per_test": 0.000607,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "openai/gpt-4.1-2025-04-14": {
                "accuracy": 91.183,
                "latency": 3.083,
                "stderr": 0.259,
                "cost_per_test": 0.002952,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "google/gemini-2.5-flash-preview-09-2025-thinking": {
                "accuracy": 91.167,
                "latency": 18.007,
                "stderr": 0.259,
                "cost_per_test": 0.001265,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "minimax/MiniMax-M2.1": {
                "accuracy": 91.158,
                "latency": 35.938,
                "stderr": 0.259,
                "cost_per_test": 0.002383,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax"
            },
            "google/gemini-2.5-flash-preview-04-17-thinking": {
                "accuracy": 91.017,
                "latency": 8.867,
                "stderr": 0.261,
                "cost_per_test": 0.003784,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "fireworks/deepseek-r1": {
                "accuracy": 90.8,
                "latency": 41.569,
                "stderr": 0.518,
                "cost_per_test": 0.005903,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "fireworks/qwen3-235b-a22b": {
                "accuracy": 90.617,
                "latency": 26.257,
                "stderr": 0.266,
                "cost_per_test": 0.001177,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "anthropic/claude-sonnet-4-20250514": {
                "accuracy": 90.35,
                "latency": 8.713,
                "stderr": 0.27,
                "cost_per_test": 0.007087,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "openai/o1-mini-2024-09-12": {
                "accuracy": 90.217,
                "latency": 5.885,
                "stderr": 0.532,
                "cost_per_test": 0.004296,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "anthropic/claude-3-7-sonnet-20250219-thinking": {
                "accuracy": 90.217,
                "latency": 15.739,
                "stderr": 0.532,
                "cost_per_test": 0.016693,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 6096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "grok/grok-3-mini-fast-high-reasoning": {
                "accuracy": 90.1,
                "latency": 7.065,
                "stderr": 0.273,
                "cost_per_test": 0.000765,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "zai/glm-4.5": {
                "accuracy": 89.975,
                "latency": 59.609,
                "stderr": 0.274,
                "cost_per_test": 0.003386,
                "temperature": 0.6,
                "top_p": 1.0,
                "max_output_tokens": 81920,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI"
            },
            "mistralai/magistral-medium-2509": {
                "accuracy": 89.467,
                "latency": 17.075,
                "stderr": 0.28,
                "cost_per_test": 0.006909,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI"
            },
            "fireworks/deepseek-v3p2": {
                "accuracy": 89.45,
                "latency": 13.297,
                "stderr": 0.28,
                "cost_per_test": 0.000234,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 20480,
                "reasoning": null,
                "reasoning_effort": "none",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "google/gemini-2.5-flash-lite-preview-09-2025-thinking": {
                "accuracy": 88.867,
                "latency": 14.822,
                "stderr": 0.287,
                "cost_per_test": 0.000167,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "grok/grok-3-mini-fast-low-reasoning": {
                "accuracy": 88.65,
                "latency": 4.883,
                "stderr": 0.29,
                "cost_per_test": 0.000558,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "low",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "together/meta-llama/Meta-Llama-3.1-405B-Instruct-Turbo": {
                "accuracy": 88.242,
                "latency": 8.747,
                "stderr": 0.576,
                "cost_per_test": 0.002624,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "openai/gpt-4o-2024-08-06": {
                "accuracy": 88.161,
                "latency": 3.392,
                "stderr": 0.578,
                "cost_per_test": 0.002787,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "alibaba/qwen3-max-preview": {
                "accuracy": 87.375,
                "latency": 19.552,
                "stderr": 0.303,
                "cost_per_test": 0.00318,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba"
            },
            "alibaba/qwen3-max": {
                "accuracy": 87.367,
                "latency": 0.0,
                "stderr": 0.303,
                "cost_per_test": 0.003172,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba"
            },
            "google/gemini-2.5-flash-preview-04-17": {
                "accuracy": 86.733,
                "latency": 2.254,
                "stderr": 0.31,
                "cost_per_test": 0.001068,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "together/meta-llama/Meta-Llama-3.1-70B-Instruct-Turbo": {
                "accuracy": 84.784,
                "latency": 4.802,
                "stderr": 0.642,
                "cost_per_test": 0.000618,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "openai/gpt-4.1-mini-2025-04-14": {
                "accuracy": 84.633,
                "latency": 1.858,
                "stderr": 0.329,
                "cost_per_test": 0.000519,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "together/moonshotai/Kimi-K2-Instruct": {
                "accuracy": 83.975,
                "latency": 10.984,
                "stderr": 0.335,
                "cost_per_test": 0.001028,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "grok/grok-3": {
                "accuracy": 83.85,
                "latency": 7.536,
                "stderr": 0.336,
                "cost_per_test": 0.006196,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "anthropic/claude-3-5-sonnet-20241022": {
                "accuracy": 83.191,
                "latency": 5.924,
                "stderr": 0.669,
                "cost_per_test": 0.005964,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "fireworks/gpt-oss-20b": {
                "accuracy": 82.875,
                "latency": 25.351,
                "stderr": 0.344,
                "cost_per_test": 0.000335,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "mistralai/magistral-small-2509": {
                "accuracy": 82.358,
                "latency": 7.973,
                "stderr": 0.348,
                "cost_per_test": 0.001729,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI"
            },
            "mistralai/mistral-large-2512": {
                "accuracy": 82.233,
                "latency": 17.31,
                "stderr": 0.349,
                "cost_per_test": 0.000771,
                "temperature": 0.065,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI"
            },
            "fireworks/deepseek-v3-0324": {
                "accuracy": 82.0,
                "latency": 12.876,
                "stderr": 0.351,
                "cost_per_test": 0.000614,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "openai/gpt-4-turbo": {
                "accuracy": 81.986,
                "latency": 8.413,
                "stderr": 0.686,
                "cost_per_test": 0.012152,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "google/gemini-2.0-flash-001": {
                "accuracy": 81.467,
                "latency": 1.941,
                "stderr": 0.695,
                "cost_per_test": 0.000114,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "fireworks/deepseek-v3": {
                "accuracy": 80.9,
                "latency": 6.816,
                "stderr": 0.359,
                "cost_per_test": 0.000578,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "cohere/command-a-03-2025": {
                "accuracy": 80.55,
                "latency": 2.908,
                "stderr": 0.361,
                "cost_per_test": 0.003459,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 8000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere"
            },
            "google/gemini-2.5-flash-lite-preview-09-2025": {
                "accuracy": 80.325,
                "latency": 5.145,
                "stderr": 0.363,
                "cost_per_test": 0.000281,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 79.567,
                "latency": 23.905,
                "stderr": 0.368,
                "cost_per_test": 0.014093,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "mistralai/mistral-medium-2505": {
                "accuracy": 78.233,
                "latency": 7.344,
                "stderr": 0.377,
                "cost_per_test": 0.000637,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI"
            },
            "together/Qwen/Qwen2.5-72B-Instruct-Turbo": {
                "accuracy": 77.395,
                "latency": 5.573,
                "stderr": 0.736,
                "cost_per_test": 0.000836,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "google/gemini-1.5-pro-002": {
                "accuracy": 76.53,
                "latency": 5.927,
                "stderr": 0.758,
                "cost_per_test": 0.002954,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "mistralai/mistral-large-2411": {
                "accuracy": 76.225,
                "latency": 4.86,
                "stderr": 0.761,
                "cost_per_test": 0.002428,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI"
            },
            "grok/grok-4-1-fast-non-reasoning": {
                "accuracy": 76.025,
                "latency": 6.552,
                "stderr": 0.39,
                "cost_per_test": 0.000165,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "grok/grok-4-fast-non-reasoning": {
                "accuracy": 75.358,
                "latency": 2.416,
                "stderr": 0.393,
                "cost_per_test": 0.000481,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "openai/gpt-4o-mini-2024-07-18": {
                "accuracy": 72.436,
                "latency": 2.322,
                "stderr": 0.799,
                "cost_per_test": 0.000157,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "mistralai/mistral-small-2503": {
                "accuracy": 69.1,
                "latency": 3.634,
                "stderr": 0.422,
                "cost_per_test": 0.000102,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI"
            },
            "openai/gpt-4.1-nano-2025-04-14": {
                "accuracy": 68.225,
                "latency": 1.416,
                "stderr": 0.425,
                "cost_per_test": 0.000154,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "ai21labs/jamba-1.5-large": {
                "accuracy": 68.108,
                "latency": 5.999,
                "stderr": 0.833,
                "cost_per_test": 0.002198,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs"
            },
            "together/meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo": {
                "accuracy": 62.614,
                "latency": 2.372,
                "stderr": 0.865,
                "cost_per_test": 0.000122,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "together/mistralai/Mixtral-8x22B-Instruct-v0.1": {
                "accuracy": 62.139,
                "latency": 8.946,
                "stderr": 0.867,
                "cost_per_test": 0.000918,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "openai/gpt-3.5-turbo": {
                "accuracy": 58.471,
                "latency": 1.589,
                "stderr": 0.881,
                "cost_per_test": 0.000424,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "mistralai/mistral-small-2402": {
                "accuracy": 56.983,
                "latency": 2.763,
                "stderr": 0.885,
                "cost_per_test": 0.00021,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI"
            },
            "ai21labs/jamba-1.5-mini": {
                "accuracy": 55.183,
                "latency": 1.143,
                "stderr": 0.889,
                "cost_per_test": 0.000147,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs"
            },
            "together/mistralai/Mixtral-8x7B-v0.1": {
                "accuracy": 53.218,
                "latency": 3.488,
                "stderr": 0.892,
                "cost_per_test": 0.00045,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "ai21labs/jamba-mini-1.6": {
                "accuracy": 52.517,
                "latency": 2.185,
                "stderr": 0.456,
                "cost_per_test": 0.000203,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs"
            },
            "together/meta-llama/Llama-4-Scout-17B-16E-Instruct": {
                "accuracy": 50.9,
                "latency": 4.802,
                "stderr": 0.456,
                "cost_per_test": 0.000326,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "ai21labs/jamba-large-1.6": {
                "accuracy": 50.7,
                "latency": 7.31,
                "stderr": 0.456,
                "cost_per_test": 0.003147,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs"
            },
            "fireworks/llama4-maverick-instruct-basic": {
                "accuracy": 43.3,
                "latency": 7.751,
                "stderr": 0.452,
                "cost_per_test": 0.000687,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "cohere/command-r-plus": {
                "accuracy": 2.651,
                "latency": 6.056,
                "stderr": 0.29,
                "cost_per_test": 0.003456,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 4000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere"
            }
        },
        "unbiased": {
            "openai/o1-2024-12-17": {
                "accuracy": 96.85,
                "latency": 10.626,
                "stderr": 0.77,
                "cost_per_test": 0.049024,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 96.8,
                "latency": 37.824,
                "stderr": 0.394,
                "cost_per_test": 0.016719,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 96.75,
                "latency": 9.872,
                "stderr": 0.396,
                "cost_per_test": 0.008166,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 96.65,
                "latency": 25.407,
                "stderr": 0.402,
                "cost_per_test": 0.017827,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 96.55,
                "latency": 21.304,
                "stderr": 0.408,
                "cost_per_test": 0.02758,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 96.4,
                "latency": 28.275,
                "stderr": 0.417,
                "cost_per_test": 0.025209,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 96.4,
                "latency": 30.304,
                "stderr": 0.417,
                "cost_per_test": 0.143366,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 96.3,
                "latency": 22.425,
                "stderr": 0.422,
                "cost_per_test": 0.002805,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "openai/o3-2025-04-16": {
                "accuracy": 96.25,
                "latency": 7.652,
                "stderr": 0.425,
                "cost_per_test": 0.00516,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 96.25,
                "latency": 17.662,
                "stderr": 0.425,
                "cost_per_test": 0.032642,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "openai/o4-mini-2025-04-16": {
                "accuracy": 96.15,
                "latency": 6.5,
                "stderr": 0.43,
                "cost_per_test": 0.002634,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 96.0,
                "latency": 22.393,
                "stderr": 0.438,
                "cost_per_test": 0.010852,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "alibaba/qwen3.5-plus-thinking": {
                "accuracy": 95.95,
                "latency": 50.996,
                "stderr": 0.441,
                "cost_per_test": 0.007273,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba"
            },
            "openai/o3-mini-2025-01-31": {
                "accuracy": 95.65,
                "latency": 8.06,
                "stderr": 0.897,
                "cost_per_test": 0.004868,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 95.3,
                "latency": 33.023,
                "stderr": 0.473,
                "cost_per_test": 0.029313,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 95.25,
                "latency": 19.348,
                "stderr": 0.476,
                "cost_per_test": 0.008533,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "openai/o1-preview-2024-09-12": {
                "accuracy": 95.05,
                "latency": 15.506,
                "stderr": 0.954,
                "cost_per_test": 0.10287,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 94.9,
                "latency": 37.995,
                "stderr": 0.492,
                "cost_per_test": 0.006332,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI"
            },
            "zai/glm-5-thinking": {
                "accuracy": 94.7,
                "latency": 76.074,
                "stderr": 0.501,
                "cost_per_test": 0.007552,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI"
            },
            "anthropic/claude-opus-4-5-20251101": {
                "accuracy": 94.3,
                "latency": 10.116,
                "stderr": 0.518,
                "cost_per_test": 0.042508,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "fireworks/deepseek-v3p2-thinking": {
                "accuracy": 94.3,
                "latency": 51.583,
                "stderr": 0.518,
                "cost_per_test": 0.000742,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 20480,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "anthropic/claude-opus-4-1-20250805-thinking": {
                "accuracy": 94.1,
                "latency": 23.243,
                "stderr": 0.527,
                "cost_per_test": 0.087267,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "zai/glm-4.7": {
                "accuracy": 94.05,
                "latency": 86.477,
                "stderr": 0.529,
                "cost_per_test": 0.004664,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI"
            },
            "together/Qwen/Qwen2.5-72B-Instruct-Turbo": {
                "accuracy": 93.95,
                "latency": 3.906,
                "stderr": 1.047,
                "cost_per_test": 0.000768,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "grok/grok-2-1212": {
                "accuracy": 93.95,
                "latency": 3.906,
                "stderr": 1.047,
                "cost_per_test": 0.003053,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 0,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 93.85,
                "latency": 16.253,
                "stderr": 0.537,
                "cost_per_test": 0.018308,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "anthropic/claude-opus-4-1-20250805": {
                "accuracy": 93.75,
                "latency": 11.46,
                "stderr": 0.541,
                "cost_per_test": 0.040952,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "anthropic/claude-sonnet-4-20250514-thinking": {
                "accuracy": 93.7,
                "latency": 25.602,
                "stderr": 0.543,
                "cost_per_test": 0.024067,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "grok/grok-4-0709": {
                "accuracy": 93.7,
                "latency": 58.874,
                "stderr": 0.543,
                "cost_per_test": 0.023165,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "anthropic/claude-opus-4-20250514": {
                "accuracy": 93.65,
                "latency": 11.544,
                "stderr": 0.545,
                "cost_per_test": 0.037113,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "google/gemini-2.5-pro-exp-03-25": {
                "accuracy": 93.6,
                "latency": 10.301,
                "stderr": 0.547,
                "cost_per_test": 0.002845,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "minimax/MiniMax-M2.5": {
                "accuracy": 93.2,
                "latency": 30.695,
                "stderr": 0.563,
                "cost_per_test": 0.002671,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax"
            },
            "openai/gpt-5-nano-2025-08-07": {
                "accuracy": 93.05,
                "latency": 25.505,
                "stderr": 0.569,
                "cost_per_test": 0.001101,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 92.75,
                "latency": 14.838,
                "stderr": 0.58,
                "cost_per_test": 0.000534,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "grok/grok-4-fast-reasoning": {
                "accuracy": 92.6,
                "latency": 5.017,
                "stderr": 0.585,
                "cost_per_test": 0.000714,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "kimi/kimi-k2-thinking": {
                "accuracy": 92.6,
                "latency": 202.862,
                "stderr": 0.585,
                "cost_per_test": 0.005435,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI"
            },
            "minimax/MiniMax-M2.1": {
                "accuracy": 92.25,
                "latency": 35.648,
                "stderr": 0.598,
                "cost_per_test": 0.002412,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax"
            },
            "zai/glm-4.6": {
                "accuracy": 92.2,
                "latency": 29.303,
                "stderr": 0.6,
                "cost_per_test": 0.00429,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 122880,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI"
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 92.2,
                "latency": 31.054,
                "stderr": 0.6,
                "cost_per_test": 0.031909,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "google/gemini-2.5-flash-preview-09-2025": {
                "accuracy": 92.15,
                "latency": 6.359,
                "stderr": 0.601,
                "cost_per_test": 0.001255,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "google/gemini-2.5-flash-preview-09-2025-thinking": {
                "accuracy": 92.05,
                "latency": 27.81,
                "stderr": 0.605,
                "cost_per_test": 0.00125,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "fireworks/gpt-oss-120b": {
                "accuracy": 91.9,
                "latency": 4.224,
                "stderr": 0.61,
                "cost_per_test": 0.000572,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "openai/gpt-4.1-2025-04-14": {
                "accuracy": 91.8,
                "latency": 3.158,
                "stderr": 0.614,
                "cost_per_test": 0.002863,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "google/gemini-2.5-flash-preview-04-17-thinking": {
                "accuracy": 91.75,
                "latency": 8.451,
                "stderr": 0.615,
                "cost_per_test": 0.003552,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "anthropic/claude-sonnet-4-20250514": {
                "accuracy": 91.6,
                "latency": 10.879,
                "stderr": 0.62,
                "cost_per_test": 0.007084,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "fireworks/deepseek-r1": {
                "accuracy": 91.55,
                "latency": 45.512,
                "stderr": 1.22,
                "cost_per_test": 0.005779,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "zai/glm-4.5": {
                "accuracy": 91.4,
                "latency": 50.723,
                "stderr": 0.627,
                "cost_per_test": 0.003328,
                "temperature": 0.6,
                "top_p": 1.0,
                "max_output_tokens": 81920,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI"
            },
            "fireworks/qwen3-235b-a22b": {
                "accuracy": 91.35,
                "latency": 25.911,
                "stderr": 0.629,
                "cost_per_test": 0.001133,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "anthropic/claude-3-7-sonnet-20250219-thinking": {
                "accuracy": 91.25,
                "latency": 15.386,
                "stderr": 1.24,
                "cost_per_test": 0.016438,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 6096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "openai/o1-mini-2024-09-12": {
                "accuracy": 91.1,
                "latency": 5.696,
                "stderr": 1.249,
                "cost_per_test": 0.004144,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "grok/grok-3-mini-fast-high-reasoning": {
                "accuracy": 90.8,
                "latency": 6.93,
                "stderr": 0.646,
                "cost_per_test": 0.000744,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "together/meta-llama/Meta-Llama-3.1-405B-Instruct-Turbo": {
                "accuracy": 90.7,
                "latency": 7.082,
                "stderr": 1.274,
                "cost_per_test": 0.002572,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "fireworks/deepseek-v3p2": {
                "accuracy": 90.7,
                "latency": 12.786,
                "stderr": 0.649,
                "cost_per_test": 0.00022,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 20480,
                "reasoning": null,
                "reasoning_effort": "none",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "mistralai/magistral-medium-2509": {
                "accuracy": 90.55,
                "latency": 16.457,
                "stderr": 0.654,
                "cost_per_test": 0.006683,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI"
            },
            "grok/grok-3-mini-fast-low-reasoning": {
                "accuracy": 89.4,
                "latency": 4.757,
                "stderr": 0.688,
                "cost_per_test": 0.000549,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "low",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "google/gemini-2.5-flash-lite-preview-09-2025-thinking": {
                "accuracy": 89.4,
                "latency": 24.976,
                "stderr": 0.688,
                "cost_per_test": 0.000162,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "openai/gpt-4o-2024-08-06": {
                "accuracy": 89.0,
                "latency": 3.282,
                "stderr": 1.372,
                "cost_per_test": 0.002733,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "alibaba/qwen3-max": {
                "accuracy": 88.6,
                "latency": 0.0,
                "stderr": 0.711,
                "cost_per_test": 0.003065,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba"
            },
            "alibaba/qwen3-max-preview": {
                "accuracy": 88.2,
                "latency": 21.182,
                "stderr": 0.721,
                "cost_per_test": 0.003035,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba"
            },
            "google/gemini-2.5-flash-preview-04-17": {
                "accuracy": 87.45,
                "latency": 2.402,
                "stderr": 0.741,
                "cost_per_test": 0.001205,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "together/meta-llama/Meta-Llama-3.1-70B-Instruct-Turbo": {
                "accuracy": 87.4,
                "latency": 3.452,
                "stderr": 1.455,
                "cost_per_test": 0.000597,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "grok/grok-3": {
                "accuracy": 85.35,
                "latency": 4.502,
                "stderr": 0.791,
                "cost_per_test": 0.005875,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "openai/gpt-4.1-mini-2025-04-14": {
                "accuracy": 85.1,
                "latency": 1.916,
                "stderr": 0.796,
                "cost_per_test": 0.000503,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "together/moonshotai/Kimi-K2-Instruct": {
                "accuracy": 85.05,
                "latency": 12.564,
                "stderr": 0.797,
                "cost_per_test": 0.001002,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "mistralai/magistral-small-2509": {
                "accuracy": 84.4,
                "latency": 7.899,
                "stderr": 0.811,
                "cost_per_test": 0.001709,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI"
            },
            "anthropic/claude-3-5-sonnet-20241022": {
                "accuracy": 84.2,
                "latency": 5.114,
                "stderr": 1.598,
                "cost_per_test": 0.005851,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "fireworks/gpt-oss-20b": {
                "accuracy": 83.9,
                "latency": 33.453,
                "stderr": 0.822,
                "cost_per_test": 0.00032,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "fireworks/deepseek-v3-0324": {
                "accuracy": 83.45,
                "latency": 13.141,
                "stderr": 0.831,
                "cost_per_test": 0.000593,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "cohere/command-a-03-2025": {
                "accuracy": 82.7,
                "latency": 2.861,
                "stderr": 0.846,
                "cost_per_test": 0.003378,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 8000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere"
            },
            "google/gemini-2.0-flash-001": {
                "accuracy": 82.65,
                "latency": 1.873,
                "stderr": 1.659,
                "cost_per_test": 0.000109,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "fireworks/deepseek-v3": {
                "accuracy": 82.3,
                "latency": 7.048,
                "stderr": 0.853,
                "cost_per_test": 0.000561,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "mistralai/mistral-large-2512": {
                "accuracy": 82.2,
                "latency": 14.75,
                "stderr": 0.855,
                "cost_per_test": 0.000726,
                "temperature": 0.065,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI"
            },
            "openai/gpt-4-turbo": {
                "accuracy": 81.4,
                "latency": 9.903,
                "stderr": 1.705,
                "cost_per_test": 0.012409,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "google/gemini-2.5-flash-lite-preview-09-2025": {
                "accuracy": 80.45,
                "latency": 6.098,
                "stderr": 0.887,
                "cost_per_test": 0.000276,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 80.25,
                "latency": 22.303,
                "stderr": 0.89,
                "cost_per_test": 0.01332,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "mistralai/mistral-medium-2505": {
                "accuracy": 79.35,
                "latency": 7.247,
                "stderr": 0.905,
                "cost_per_test": 0.00062,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI"
            },
            "google/gemini-1.5-pro-002": {
                "accuracy": 78.4,
                "latency": 5.692,
                "stderr": 1.803,
                "cost_per_test": 0.002873,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "grok/grok-4-1-fast-non-reasoning": {
                "accuracy": 77.5,
                "latency": 4.928,
                "stderr": 0.934,
                "cost_per_test": 0.000159,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "grok/grok-4-fast-non-reasoning": {
                "accuracy": 77.1,
                "latency": 2.283,
                "stderr": 0.94,
                "cost_per_test": 0.000467,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "mistralai/mistral-large-2411": {
                "accuracy": 76.95,
                "latency": 4.722,
                "stderr": 1.845,
                "cost_per_test": 0.00232,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI"
            },
            "openai/gpt-4o-mini-2024-07-18": {
                "accuracy": 73.3,
                "latency": 2.583,
                "stderr": 1.938,
                "cost_per_test": 0.000154,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "mistralai/mistral-small-2503": {
                "accuracy": 70.0,
                "latency": 3.468,
                "stderr": 1.025,
                "cost_per_test": 9.9e-05,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI"
            },
            "ai21labs/jamba-1.5-large": {
                "accuracy": 69.65,
                "latency": 7.963,
                "stderr": 2.013,
                "cost_per_test": 0.002182,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs"
            },
            "openai/gpt-4.1-nano-2025-04-14": {
                "accuracy": 69.3,
                "latency": 1.37,
                "stderr": 1.031,
                "cost_per_test": 0.000151,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "together/meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo": {
                "accuracy": 63.8,
                "latency": 2.571,
                "stderr": 2.104,
                "cost_per_test": 0.000118,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "together/mistralai/Mixtral-8x22B-Instruct-v0.1": {
                "accuracy": 62.9,
                "latency": 10.923,
                "stderr": 2.115,
                "cost_per_test": 0.000888,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "openai/gpt-3.5-turbo": {
                "accuracy": 59.25,
                "latency": 1.588,
                "stderr": 2.152,
                "cost_per_test": 0.000415,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "mistralai/mistral-small-2402": {
                "accuracy": 58.0,
                "latency": 2.784,
                "stderr": 2.161,
                "cost_per_test": 0.000205,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI"
            },
            "ai21labs/jamba-1.5-mini": {
                "accuracy": 56.45,
                "latency": 1.106,
                "stderr": 2.171,
                "cost_per_test": 0.000143,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs"
            },
            "together/meta-llama/Llama-4-Scout-17B-16E-Instruct": {
                "accuracy": 54.4,
                "latency": 4.58,
                "stderr": 1.114,
                "cost_per_test": 0.000319,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "together/mistralai/Mixtral-8x7B-v0.1": {
                "accuracy": 53.75,
                "latency": 4.475,
                "stderr": 2.183,
                "cost_per_test": 0.000434,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "ai21labs/jamba-mini-1.6": {
                "accuracy": 52.9,
                "latency": 2.192,
                "stderr": 1.116,
                "cost_per_test": 0.000198,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs"
            },
            "ai21labs/jamba-large-1.6": {
                "accuracy": 50.7,
                "latency": 6.962,
                "stderr": 1.118,
                "cost_per_test": 0.003101,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs"
            },
            "fireworks/llama4-maverick-instruct-basic": {
                "accuracy": 47.8,
                "latency": 7.791,
                "stderr": 1.117,
                "cost_per_test": 0.000683,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "cohere/command-r-plus": {
                "accuracy": 3.0,
                "latency": 5.645,
                "stderr": 0.752,
                "cost_per_test": 0.003396,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 4000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere"
            }
        },
        "hispanic": {
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 96.55,
                "latency": 14.587,
                "stderr": 0.408,
                "cost_per_test": 0.010282,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "openai/o1-2024-12-17": {
                "accuracy": 96.45,
                "latency": 11.213,
                "stderr": 0.815,
                "cost_per_test": 0.053826,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "openai/o3-2025-04-16": {
                "accuracy": 96.15,
                "latency": 9.158,
                "stderr": 0.43,
                "cost_per_test": 0.005471,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 96.05,
                "latency": 40.792,
                "stderr": 0.436,
                "cost_per_test": 0.022021,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 96.0,
                "latency": 9.225,
                "stderr": 0.438,
                "cost_per_test": 0.002937,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 95.95,
                "latency": 38.535,
                "stderr": 0.441,
                "cost_per_test": 0.018084,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "openai/o4-mini-2025-04-16": {
                "accuracy": 95.9,
                "latency": 6.819,
                "stderr": 0.443,
                "cost_per_test": 0.002826,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 95.7,
                "latency": 23.871,
                "stderr": 0.454,
                "cost_per_test": 0.045987,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 95.45,
                "latency": 22.971,
                "stderr": 0.466,
                "cost_per_test": 0.029478,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 95.4,
                "latency": 26.799,
                "stderr": 0.468,
                "cost_per_test": 0.012929,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 95.152,
                "latency": 31.523,
                "stderr": 0.48,
                "cost_per_test": 0.146874,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "alibaba/qwen3.5-plus-thinking": {
                "accuracy": 94.85,
                "latency": 44.226,
                "stderr": 0.494,
                "cost_per_test": 0.008501,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba"
            },
            "openai/o3-mini-2025-01-31": {
                "accuracy": 94.5,
                "latency": 8.469,
                "stderr": 1.002,
                "cost_per_test": 0.005409,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 94.4,
                "latency": 30.934,
                "stderr": 0.514,
                "cost_per_test": 0.027236,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 94.35,
                "latency": 20.4,
                "stderr": 0.516,
                "cost_per_test": 0.022449,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 94.1,
                "latency": 8.629,
                "stderr": 0.527,
                "cost_per_test": 0.009188,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 94.1,
                "latency": 36.464,
                "stderr": 0.527,
                "cost_per_test": 0.031817,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 94.1,
                "latency": 42.99,
                "stderr": 0.527,
                "cost_per_test": 0.007231,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI"
            },
            "zai/glm-5-thinking": {
                "accuracy": 93.95,
                "latency": 83.77,
                "stderr": 0.533,
                "cost_per_test": 0.008235,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI"
            },
            "fireworks/deepseek-v3p2-thinking": {
                "accuracy": 93.85,
                "latency": 54.748,
                "stderr": 0.537,
                "cost_per_test": 0.000787,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 20480,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "openai/o1-preview-2024-09-12": {
                "accuracy": 93.7,
                "latency": 16.142,
                "stderr": 1.067,
                "cost_per_test": 0.11043,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "anthropic/claude-opus-4-1-20250805-thinking": {
                "accuracy": 93.55,
                "latency": 23.615,
                "stderr": 0.549,
                "cost_per_test": 0.088116,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "zai/glm-4.7": {
                "accuracy": 93.45,
                "latency": 55.079,
                "stderr": 0.553,
                "cost_per_test": 0.004902,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI"
            },
            "openai/gpt-5-nano-2025-08-07": {
                "accuracy": 93.3,
                "latency": 19.947,
                "stderr": 0.559,
                "cost_per_test": 0.001289,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "anthropic/claude-opus-4-5-20251101": {
                "accuracy": 92.95,
                "latency": 10.452,
                "stderr": 0.572,
                "cost_per_test": 0.043333,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "kimi/kimi-k2-thinking": {
                "accuracy": 92.9,
                "latency": 228.356,
                "stderr": 0.574,
                "cost_per_test": 0.006258,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI"
            },
            "anthropic/claude-opus-4-1-20250805": {
                "accuracy": 92.65,
                "latency": 11.688,
                "stderr": 0.584,
                "cost_per_test": 0.041511,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "anthropic/claude-opus-4-20250514": {
                "accuracy": 92.4,
                "latency": 12.246,
                "stderr": 0.593,
                "cost_per_test": 0.038091,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "minimax/MiniMax-M2.5": {
                "accuracy": 92.3,
                "latency": 34.61,
                "stderr": 0.596,
                "cost_per_test": 0.003005,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax"
            },
            "google/gemini-2.5-pro-exp-03-25": {
                "accuracy": 92.25,
                "latency": 10.852,
                "stderr": 0.598,
                "cost_per_test": 0.002918,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "zai/glm-4.6": {
                "accuracy": 92.0,
                "latency": 28.316,
                "stderr": 0.607,
                "cost_per_test": 0.00421,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 122880,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI"
            },
            "grok/grok-4-0709": {
                "accuracy": 92.0,
                "latency": 65.196,
                "stderr": 0.607,
                "cost_per_test": 0.026197,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "grok/grok-4-fast-reasoning": {
                "accuracy": 91.95,
                "latency": 5.329,
                "stderr": 0.608,
                "cost_per_test": 0.000746,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "anthropic/claude-sonnet-4-20250514-thinking": {
                "accuracy": 91.9,
                "latency": 27.152,
                "stderr": 0.61,
                "cost_per_test": 0.026032,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 91.9,
                "latency": 35.094,
                "stderr": 0.61,
                "cost_per_test": 0.035386,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 91.75,
                "latency": 13.495,
                "stderr": 0.615,
                "cost_per_test": 0.000578,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "grok/grok-2-1212": {
                "accuracy": 91.7,
                "latency": 4.109,
                "stderr": 1.211,
                "cost_per_test": 0.003321,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 0,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "fireworks/gpt-oss-120b": {
                "accuracy": 91.15,
                "latency": 5.397,
                "stderr": 0.635,
                "cost_per_test": 0.000621,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "google/gemini-2.5-flash-preview-09-2025-thinking": {
                "accuracy": 91.15,
                "latency": 10.276,
                "stderr": 0.635,
                "cost_per_test": 0.001268,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "minimax/MiniMax-M2.1": {
                "accuracy": 91.1,
                "latency": 36.786,
                "stderr": 0.637,
                "cost_per_test": 0.002401,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax"
            },
            "openai/gpt-4.1-2025-04-14": {
                "accuracy": 91.0,
                "latency": 2.929,
                "stderr": 0.64,
                "cost_per_test": 0.00295,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "google/gemini-2.5-flash-preview-04-17-thinking": {
                "accuracy": 90.9,
                "latency": 9.084,
                "stderr": 0.643,
                "cost_per_test": 0.003874,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "grok/grok-3-mini-fast-high-reasoning": {
                "accuracy": 90.65,
                "latency": 6.997,
                "stderr": 0.651,
                "cost_per_test": 0.000761,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "google/gemini-2.5-flash-preview-09-2025": {
                "accuracy": 90.55,
                "latency": 18.656,
                "stderr": 0.654,
                "cost_per_test": 0.001271,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "fireworks/deepseek-r1": {
                "accuracy": 90.55,
                "latency": 40.14,
                "stderr": 1.283,
                "cost_per_test": 0.005998,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "openai/o1-mini-2024-09-12": {
                "accuracy": 90.3,
                "latency": 5.98,
                "stderr": 1.298,
                "cost_per_test": 0.004346,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "anthropic/claude-3-7-sonnet-20250219-thinking": {
                "accuracy": 90.25,
                "latency": 15.694,
                "stderr": 1.301,
                "cost_per_test": 0.01674,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 6096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "fireworks/qwen3-235b-a22b": {
                "accuracy": 90.25,
                "latency": 26.426,
                "stderr": 0.663,
                "cost_per_test": 0.001192,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "anthropic/claude-sonnet-4-20250514": {
                "accuracy": 89.8,
                "latency": 7.699,
                "stderr": 0.677,
                "cost_per_test": 0.007091,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "zai/glm-4.5": {
                "accuracy": 89.8,
                "latency": 59.036,
                "stderr": 0.677,
                "cost_per_test": 0.003343,
                "temperature": 0.6,
                "top_p": 1.0,
                "max_output_tokens": 81920,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI"
            },
            "mistralai/magistral-medium-2509": {
                "accuracy": 89.5,
                "latency": 17.2,
                "stderr": 0.686,
                "cost_per_test": 0.006951,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI"
            },
            "fireworks/deepseek-v3p2": {
                "accuracy": 88.9,
                "latency": 13.622,
                "stderr": 0.702,
                "cost_per_test": 0.000239,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 20480,
                "reasoning": null,
                "reasoning_effort": "none",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "google/gemini-2.5-flash-lite-preview-09-2025-thinking": {
                "accuracy": 88.8,
                "latency": 23.509,
                "stderr": 0.705,
                "cost_per_test": 0.000166,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "grok/grok-3-mini-fast-low-reasoning": {
                "accuracy": 88.3,
                "latency": 4.956,
                "stderr": 0.719,
                "cost_per_test": 0.000557,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "low",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "together/meta-llama/Meta-Llama-3.1-405B-Instruct-Turbo": {
                "accuracy": 88.15,
                "latency": 9.337,
                "stderr": 1.417,
                "cost_per_test": 0.002677,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "openai/gpt-4o-2024-08-06": {
                "accuracy": 87.9,
                "latency": 3.374,
                "stderr": 1.43,
                "cost_per_test": 0.002799,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "alibaba/qwen3-max": {
                "accuracy": 86.9,
                "latency": 0.0,
                "stderr": 0.754,
                "cost_per_test": 0.003205,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba"
            },
            "alibaba/qwen3-max-preview": {
                "accuracy": 86.7,
                "latency": 17.982,
                "stderr": 0.759,
                "cost_per_test": 0.003193,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba"
            },
            "google/gemini-2.5-flash-preview-04-17": {
                "accuracy": 86.4,
                "latency": 2.256,
                "stderr": 0.766,
                "cost_per_test": 0.001069,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "openai/gpt-4.1-mini-2025-04-14": {
                "accuracy": 84.8,
                "latency": 1.833,
                "stderr": 0.803,
                "cost_per_test": 0.000518,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "together/meta-llama/Meta-Llama-3.1-70B-Instruct-Turbo": {
                "accuracy": 83.9,
                "latency": 3.994,
                "stderr": 1.61,
                "cost_per_test": 0.000624,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "together/moonshotai/Kimi-K2-Instruct": {
                "accuracy": 83.55,
                "latency": 7.618,
                "stderr": 0.829,
                "cost_per_test": 0.001031,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "anthropic/claude-3-5-sonnet-20241022": {
                "accuracy": 83.25,
                "latency": 5.424,
                "stderr": 1.636,
                "cost_per_test": 0.00598,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "grok/grok-3": {
                "accuracy": 83.2,
                "latency": 7.415,
                "stderr": 0.836,
                "cost_per_test": 0.006224,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "fireworks/gpt-oss-20b": {
                "accuracy": 83.1,
                "latency": 27.009,
                "stderr": 0.838,
                "cost_per_test": 0.000342,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "mistralai/mistral-large-2512": {
                "accuracy": 82.75,
                "latency": 17.954,
                "stderr": 0.845,
                "cost_per_test": 0.000768,
                "temperature": 0.065,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI"
            },
            "fireworks/deepseek-v3-0324": {
                "accuracy": 81.45,
                "latency": 11.758,
                "stderr": 0.869,
                "cost_per_test": 0.000618,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "mistralai/magistral-small-2509": {
                "accuracy": 81.35,
                "latency": 7.923,
                "stderr": 0.871,
                "cost_per_test": 0.001742,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI"
            },
            "google/gemini-2.0-flash-001": {
                "accuracy": 81.0,
                "latency": 1.979,
                "stderr": 1.719,
                "cost_per_test": 0.000115,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "fireworks/deepseek-v3": {
                "accuracy": 80.55,
                "latency": 6.738,
                "stderr": 0.885,
                "cost_per_test": 0.000581,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "openai/gpt-4-turbo": {
                "accuracy": 80.5,
                "latency": 9.181,
                "stderr": 1.736,
                "cost_per_test": 0.01274,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 80.45,
                "latency": 24.13,
                "stderr": 0.887,
                "cost_per_test": 0.014138,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "cohere/command-a-03-2025": {
                "accuracy": 79.7,
                "latency": 2.911,
                "stderr": 0.899,
                "cost_per_test": 0.003475,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 8000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere"
            },
            "google/gemini-2.5-flash-lite-preview-09-2025": {
                "accuracy": 79.2,
                "latency": 4.741,
                "stderr": 0.908,
                "cost_per_test": 0.000285,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "mistralai/mistral-medium-2505": {
                "accuracy": 77.3,
                "latency": 7.291,
                "stderr": 0.937,
                "cost_per_test": 0.000636,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI"
            },
            "mistralai/mistral-large-2411": {
                "accuracy": 76.0,
                "latency": 4.796,
                "stderr": 1.871,
                "cost_per_test": 0.002442,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI"
            },
            "grok/grok-4-1-fast-non-reasoning": {
                "accuracy": 75.45,
                "latency": 7.301,
                "stderr": 0.962,
                "cost_per_test": 0.000166,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "google/gemini-1.5-pro-002": {
                "accuracy": 75.35,
                "latency": 5.938,
                "stderr": 1.888,
                "cost_per_test": 0.002973,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "grok/grok-4-fast-non-reasoning": {
                "accuracy": 75.25,
                "latency": 2.396,
                "stderr": 0.965,
                "cost_per_test": 0.000483,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "together/Qwen/Qwen2.5-72B-Instruct-Turbo": {
                "accuracy": 73.9,
                "latency": 6.154,
                "stderr": 1.924,
                "cost_per_test": 0.00085,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "openai/gpt-4o-mini-2024-07-18": {
                "accuracy": 72.3,
                "latency": 2.477,
                "stderr": 1.96,
                "cost_per_test": 0.000157,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "mistralai/mistral-small-2503": {
                "accuracy": 68.25,
                "latency": 3.678,
                "stderr": 1.041,
                "cost_per_test": 0.000103,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI"
            },
            "ai21labs/jamba-1.5-large": {
                "accuracy": 68.15,
                "latency": 3.931,
                "stderr": 2.04,
                "cost_per_test": 0.002195,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs"
            },
            "openai/gpt-4.1-nano-2025-04-14": {
                "accuracy": 67.7,
                "latency": 1.368,
                "stderr": 1.046,
                "cost_per_test": 0.000155,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "together/mistralai/Mixtral-8x22B-Instruct-v0.1": {
                "accuracy": 62.0,
                "latency": 6.879,
                "stderr": 2.125,
                "cost_per_test": 0.000923,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "together/meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo": {
                "accuracy": 61.35,
                "latency": 3.235,
                "stderr": 2.132,
                "cost_per_test": 0.000123,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "openai/gpt-3.5-turbo": {
                "accuracy": 58.0,
                "latency": 1.495,
                "stderr": 2.161,
                "cost_per_test": 0.000426,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "mistralai/mistral-small-2402": {
                "accuracy": 56.5,
                "latency": 2.743,
                "stderr": 2.171,
                "cost_per_test": 0.000211,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI"
            },
            "ai21labs/jamba-1.5-mini": {
                "accuracy": 55.05,
                "latency": 1.16,
                "stderr": 2.178,
                "cost_per_test": 0.000147,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs"
            },
            "together/mistralai/Mixtral-8x7B-v0.1": {
                "accuracy": 53.0,
                "latency": 3.902,
                "stderr": 2.185,
                "cost_per_test": 0.000452,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "ai21labs/jamba-mini-1.6": {
                "accuracy": 51.95,
                "latency": 2.196,
                "stderr": 1.117,
                "cost_per_test": 0.000204,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs"
            },
            "ai21labs/jamba-large-1.6": {
                "accuracy": 50.7,
                "latency": 6.784,
                "stderr": 1.118,
                "cost_per_test": 0.003139,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs"
            },
            "together/meta-llama/Llama-4-Scout-17B-16E-Instruct": {
                "accuracy": 48.9,
                "latency": 4.668,
                "stderr": 1.118,
                "cost_per_test": 0.000328,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "fireworks/llama4-maverick-instruct-basic": {
                "accuracy": 41.65,
                "latency": 7.708,
                "stderr": 1.102,
                "cost_per_test": 0.000677,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "cohere/command-r-plus": {
                "accuracy": 2.45,
                "latency": 6.995,
                "stderr": 0.683,
                "cost_per_test": 0.003442,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 4000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere"
            }
        },
        "black": {
            "google/gemini-3.1-pro-preview": {
                "accuracy": 96.8,
                "latency": 39.836,
                "stderr": 0.394,
                "cost_per_test": 0.018058,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "openai/o1-2024-12-17": {
                "accuracy": 96.6,
                "latency": 11.024,
                "stderr": 0.798,
                "cost_per_test": 0.054686,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 96.5,
                "latency": 23.691,
                "stderr": 0.411,
                "cost_per_test": 0.045765,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 96.5,
                "latency": 40.726,
                "stderr": 0.411,
                "cost_per_test": 0.022269,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 96.25,
                "latency": 24.581,
                "stderr": 0.425,
                "cost_per_test": 0.003001,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 96.15,
                "latency": 14.392,
                "stderr": 0.43,
                "cost_per_test": 0.010333,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "openai/o4-mini-2025-04-16": {
                "accuracy": 96.1,
                "latency": 6.473,
                "stderr": 0.433,
                "cost_per_test": 0.002808,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 95.95,
                "latency": 32.046,
                "stderr": 0.441,
                "cost_per_test": 0.14927,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 95.85,
                "latency": 21.0,
                "stderr": 0.446,
                "cost_per_test": 0.02927,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "openai/o3-2025-04-16": {
                "accuracy": 95.75,
                "latency": 8.003,
                "stderr": 0.451,
                "cost_per_test": 0.005553,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 95.7,
                "latency": 27.728,
                "stderr": 0.454,
                "cost_per_test": 0.013337,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "alibaba/qwen3.5-plus-thinking": {
                "accuracy": 95.7,
                "latency": 58.876,
                "stderr": 0.454,
                "cost_per_test": 0.008654,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba"
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 95.25,
                "latency": 29.991,
                "stderr": 0.476,
                "cost_per_test": 0.026857,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "openai/o3-mini-2025-01-31": {
                "accuracy": 94.75,
                "latency": 8.338,
                "stderr": 0.98,
                "cost_per_test": 0.005435,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 94.75,
                "latency": 20.022,
                "stderr": 0.499,
                "cost_per_test": 0.022539,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 94.7,
                "latency": 36.422,
                "stderr": 0.501,
                "cost_per_test": 0.031987,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 94.5,
                "latency": 8.869,
                "stderr": 0.51,
                "cost_per_test": 0.009265,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "zai/glm-5-thinking": {
                "accuracy": 94.4,
                "latency": 81.957,
                "stderr": 0.514,
                "cost_per_test": 0.008175,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI"
            },
            "zai/glm-4.7": {
                "accuracy": 94.3,
                "latency": 91.035,
                "stderr": 0.518,
                "cost_per_test": 0.004844,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI"
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 94.15,
                "latency": 42.738,
                "stderr": 0.525,
                "cost_per_test": 0.007194,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI"
            },
            "fireworks/deepseek-v3p2-thinking": {
                "accuracy": 94.05,
                "latency": 54.08,
                "stderr": 0.529,
                "cost_per_test": 0.000779,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 20480,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "anthropic/claude-opus-4-1-20250805-thinking": {
                "accuracy": 93.8,
                "latency": 23.781,
                "stderr": 0.539,
                "cost_per_test": 0.088546,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "openai/o1-preview-2024-09-12": {
                "accuracy": 93.6,
                "latency": 16.288,
                "stderr": 1.075,
                "cost_per_test": 0.112495,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "openai/gpt-5-nano-2025-08-07": {
                "accuracy": 93.2,
                "latency": 15.351,
                "stderr": 0.563,
                "cost_per_test": 0.001324,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "google/gemini-2.5-pro-exp-03-25": {
                "accuracy": 93.1,
                "latency": 10.803,
                "stderr": 0.567,
                "cost_per_test": 0.002909,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "kimi/kimi-k2-thinking": {
                "accuracy": 92.8,
                "latency": 240.666,
                "stderr": 0.578,
                "cost_per_test": 0.006583,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI"
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 92.75,
                "latency": 36.466,
                "stderr": 0.58,
                "cost_per_test": 0.036805,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "anthropic/claude-opus-4-5-20251101": {
                "accuracy": 92.7,
                "latency": 10.429,
                "stderr": 0.582,
                "cost_per_test": 0.043374,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "anthropic/claude-sonnet-4-20250514-thinking": {
                "accuracy": 92.55,
                "latency": 27.996,
                "stderr": 0.587,
                "cost_per_test": 0.025562,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 92.3,
                "latency": 14.992,
                "stderr": 0.596,
                "cost_per_test": 0.000589,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "anthropic/claude-opus-4-20250514": {
                "accuracy": 92.25,
                "latency": 11.811,
                "stderr": 0.598,
                "cost_per_test": 0.03787,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "grok/grok-2-1212": {
                "accuracy": 92.2,
                "latency": 4.204,
                "stderr": 1.177,
                "cost_per_test": 0.003354,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 0,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "minimax/MiniMax-M2.5": {
                "accuracy": 92.1,
                "latency": 34.785,
                "stderr": 0.603,
                "cost_per_test": 0.003019,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax"
            },
            "grok/grok-4-fast-reasoning": {
                "accuracy": 91.9,
                "latency": 5.464,
                "stderr": 0.61,
                "cost_per_test": 0.000762,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "grok/grok-4-0709": {
                "accuracy": 91.85,
                "latency": 67.636,
                "stderr": 0.612,
                "cost_per_test": 0.027553,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "anthropic/claude-opus-4-1-20250805": {
                "accuracy": 91.8,
                "latency": 11.573,
                "stderr": 0.614,
                "cost_per_test": 0.041508,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "zai/glm-4.6": {
                "accuracy": 91.8,
                "latency": 28.772,
                "stderr": 0.614,
                "cost_per_test": 0.0043,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 122880,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI"
            },
            "fireworks/deepseek-r1": {
                "accuracy": 91.2,
                "latency": 40.196,
                "stderr": 1.243,
                "cost_per_test": 0.005973,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "minimax/MiniMax-M2.1": {
                "accuracy": 91.15,
                "latency": 36.27,
                "stderr": 0.635,
                "cost_per_test": 0.00238,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax"
            },
            "google/gemini-2.5-flash-preview-09-2025": {
                "accuracy": 90.9,
                "latency": 6.653,
                "stderr": 0.643,
                "cost_per_test": 0.001285,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "fireworks/gpt-oss-120b": {
                "accuracy": 90.809,
                "latency": 4.62,
                "stderr": 0.341,
                "cost_per_test": 0.000626,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "openai/gpt-4.1-2025-04-14": {
                "accuracy": 90.8,
                "latency": 3.196,
                "stderr": 0.646,
                "cost_per_test": 0.002987,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "fireworks/qwen3-235b-a22b": {
                "accuracy": 90.8,
                "latency": 26.989,
                "stderr": 0.646,
                "cost_per_test": 0.00118,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "google/gemini-2.5-flash-preview-09-2025-thinking": {
                "accuracy": 90.7,
                "latency": 10.489,
                "stderr": 0.649,
                "cost_per_test": 0.001274,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "google/gemini-2.5-flash-preview-04-17-thinking": {
                "accuracy": 90.5,
                "latency": 9.027,
                "stderr": 0.656,
                "cost_per_test": 0.003812,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "anthropic/claude-3-7-sonnet-20250219-thinking": {
                "accuracy": 90.1,
                "latency": 15.918,
                "stderr": 1.31,
                "cost_per_test": 0.016899,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 6096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "anthropic/claude-sonnet-4-20250514": {
                "accuracy": 90.0,
                "latency": 11.088,
                "stderr": 0.671,
                "cost_per_test": 0.007082,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "zai/glm-4.5": {
                "accuracy": 89.95,
                "latency": 46.855,
                "stderr": 0.672,
                "cost_per_test": 0.003401,
                "temperature": 0.6,
                "top_p": 1.0,
                "max_output_tokens": 81920,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI"
            },
            "grok/grok-3-mini-fast-high-reasoning": {
                "accuracy": 89.7,
                "latency": 7.184,
                "stderr": 0.68,
                "cost_per_test": 0.00078,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "openai/o1-mini-2024-09-12": {
                "accuracy": 89.65,
                "latency": 5.909,
                "stderr": 1.336,
                "cost_per_test": 0.00433,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "fireworks/deepseek-v3p2": {
                "accuracy": 89.65,
                "latency": 13.633,
                "stderr": 0.681,
                "cost_per_test": 0.000242,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 20480,
                "reasoning": null,
                "reasoning_effort": "none",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "mistralai/magistral-medium-2509": {
                "accuracy": 89.25,
                "latency": 17.103,
                "stderr": 0.693,
                "cost_per_test": 0.007058,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI"
            },
            "grok/grok-3-mini-fast-low-reasoning": {
                "accuracy": 89.05,
                "latency": 4.894,
                "stderr": 0.698,
                "cost_per_test": 0.000564,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "low",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "google/gemini-2.5-flash-lite-preview-09-2025-thinking": {
                "accuracy": 88.65,
                "latency": 6.805,
                "stderr": 0.709,
                "cost_per_test": 0.000169,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "openai/gpt-4o-2024-08-06": {
                "accuracy": 87.813,
                "latency": 3.14,
                "stderr": 1.436,
                "cost_per_test": 0.002807,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "openai/gpt-4-turbo": {
                "accuracy": 87.813,
                "latency": 3.14,
                "stderr": 1.436,
                "cost_per_test": 0.009554,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "alibaba/qwen3-max": {
                "accuracy": 87.35,
                "latency": 0.0,
                "stderr": 0.743,
                "cost_per_test": 0.00319,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba"
            },
            "alibaba/qwen3-max-preview": {
                "accuracy": 87.35,
                "latency": 18.076,
                "stderr": 0.743,
                "cost_per_test": 0.003208,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba"
            },
            "google/gemini-2.5-flash-preview-04-17": {
                "accuracy": 86.7,
                "latency": 2.428,
                "stderr": 0.759,
                "cost_per_test": 0.001156,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "openai/gpt-4.1-mini-2025-04-14": {
                "accuracy": 84.9,
                "latency": 1.818,
                "stderr": 0.801,
                "cost_per_test": 0.000517,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "together/meta-llama/Meta-Llama-3.1-70B-Instruct-Turbo": {
                "accuracy": 84.403,
                "latency": 4.329,
                "stderr": 1.592,
                "cost_per_test": 0.000625,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "together/meta-llama/Meta-Llama-3.1-405B-Instruct-Turbo": {
                "accuracy": 84.403,
                "latency": 4.329,
                "stderr": 1.592,
                "cost_per_test": 0.002486,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "grok/grok-3": {
                "accuracy": 84.3,
                "latency": 8.077,
                "stderr": 0.814,
                "cost_per_test": 0.006245,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "together/moonshotai/Kimi-K2-Instruct": {
                "accuracy": 84.2,
                "latency": 12.786,
                "stderr": 0.816,
                "cost_per_test": 0.001039,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "fireworks/gpt-oss-20b": {
                "accuracy": 83.45,
                "latency": 30.385,
                "stderr": 0.831,
                "cost_per_test": 0.000336,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "anthropic/claude-3-5-sonnet-20241022": {
                "accuracy": 82.899,
                "latency": 6.792,
                "stderr": 1.652,
                "cost_per_test": 0.006014,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "mistralai/mistral-large-2512": {
                "accuracy": 81.85,
                "latency": 17.944,
                "stderr": 0.862,
                "cost_per_test": 0.000783,
                "temperature": 0.065,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI"
            },
            "fireworks/deepseek-v3-0324": {
                "accuracy": 81.4,
                "latency": 13.065,
                "stderr": 0.87,
                "cost_per_test": 0.000621,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "google/gemini-2.5-flash-lite-preview-09-2025": {
                "accuracy": 81.15,
                "latency": 5.95,
                "stderr": 0.875,
                "cost_per_test": 0.000283,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "mistralai/magistral-small-2509": {
                "accuracy": 81.1,
                "latency": 7.957,
                "stderr": 0.875,
                "cost_per_test": 0.001745,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI"
            },
            "fireworks/deepseek-v3": {
                "accuracy": 80.9,
                "latency": 7.161,
                "stderr": 0.879,
                "cost_per_test": 0.000581,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "google/gemini-2.0-flash-001": {
                "accuracy": 80.45,
                "latency": 1.933,
                "stderr": 1.737,
                "cost_per_test": 0.000115,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 80.25,
                "latency": 24.556,
                "stderr": 0.89,
                "cost_per_test": 0.01447,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "cohere/command-a-03-2025": {
                "accuracy": 79.8,
                "latency": 2.919,
                "stderr": 0.898,
                "cost_per_test": 0.003489,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 8000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere"
            },
            "mistralai/mistral-medium-2505": {
                "accuracy": 77.95,
                "latency": 7.484,
                "stderr": 0.927,
                "cost_per_test": 0.00064,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI"
            },
            "google/gemini-1.5-pro-002": {
                "accuracy": 76.128,
                "latency": 5.905,
                "stderr": 1.87,
                "cost_per_test": 0.002995,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "mistralai/mistral-large-2411": {
                "accuracy": 75.8,
                "latency": 4.87,
                "stderr": 1.876,
                "cost_per_test": 0.002446,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI"
            },
            "grok/grok-4-1-fast-non-reasoning": {
                "accuracy": 75.55,
                "latency": 5.003,
                "stderr": 0.961,
                "cost_per_test": 0.000168,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "grok/grok-4-fast-non-reasoning": {
                "accuracy": 74.75,
                "latency": 2.398,
                "stderr": 0.972,
                "cost_per_test": 0.000484,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "together/Qwen/Qwen2.5-72B-Instruct-Turbo": {
                "accuracy": 73.671,
                "latency": 5.778,
                "stderr": 1.932,
                "cost_per_test": 0.000853,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "openai/gpt-4o-mini-2024-07-18": {
                "accuracy": 72.467,
                "latency": 2.267,
                "stderr": 1.959,
                "cost_per_test": 0.000157,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "mistralai/mistral-small-2503": {
                "accuracy": 68.25,
                "latency": 3.579,
                "stderr": 1.041,
                "cost_per_test": 0.000104,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI"
            },
            "openai/gpt-4.1-nano-2025-04-14": {
                "accuracy": 67.8,
                "latency": 1.385,
                "stderr": 1.045,
                "cost_per_test": 0.000156,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "ai21labs/jamba-1.5-large": {
                "accuracy": 67.6,
                "latency": 4.152,
                "stderr": 2.049,
                "cost_per_test": 0.002206,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs"
            },
            "together/meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo": {
                "accuracy": 62.237,
                "latency": 2.947,
                "stderr": 2.126,
                "cost_per_test": 0.000123,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "together/mistralai/Mixtral-8x22B-Instruct-v0.1": {
                "accuracy": 61.585,
                "latency": 6.535,
                "stderr": 2.133,
                "cost_per_test": 0.000927,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "openai/gpt-3.5-turbo": {
                "accuracy": 58.726,
                "latency": 1.771,
                "stderr": 2.159,
                "cost_per_test": 0.000428,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "mistralai/mistral-small-2402": {
                "accuracy": 56.65,
                "latency": 2.744,
                "stderr": 2.17,
                "cost_per_test": 0.000211,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI"
            },
            "ai21labs/jamba-1.5-mini": {
                "accuracy": 54.4,
                "latency": 1.195,
                "stderr": 2.181,
                "cost_per_test": 0.000149,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs"
            },
            "together/mistralai/Mixtral-8x7B-v0.1": {
                "accuracy": 52.959,
                "latency": 3.933,
                "stderr": 2.189,
                "cost_per_test": 0.000455,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "ai21labs/jamba-mini-1.6": {
                "accuracy": 51.15,
                "latency": 2.153,
                "stderr": 1.118,
                "cost_per_test": 0.000205,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs"
            },
            "ai21labs/jamba-large-1.6": {
                "accuracy": 50.35,
                "latency": 7.502,
                "stderr": 1.118,
                "cost_per_test": 0.003163,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs"
            },
            "together/meta-llama/Llama-4-Scout-17B-16E-Instruct": {
                "accuracy": 50.2,
                "latency": 4.936,
                "stderr": 1.118,
                "cost_per_test": 0.000328,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "fireworks/llama4-maverick-instruct-basic": {
                "accuracy": 43.7,
                "latency": 7.652,
                "stderr": 1.109,
                "cost_per_test": 0.000672,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "cohere/command-r-plus": {
                "accuracy": 2.257,
                "latency": 6.719,
                "stderr": 0.658,
                "cost_per_test": 0.003487,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 4000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere"
            }
        },
        "asian": {
            "openai/o1-2024-12-17": {
                "accuracy": 96.4,
                "latency": 10.609,
                "stderr": 0.82,
                "cost_per_test": 0.052854,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "openai/o3-2025-04-16": {
                "accuracy": 96.35,
                "latency": 8.649,
                "stderr": 0.419,
                "cost_per_test": 0.005384,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 96.3,
                "latency": 21.363,
                "stderr": 0.422,
                "cost_per_test": 0.029147,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 96.3,
                "latency": 29.586,
                "stderr": 0.422,
                "cost_per_test": 0.02141,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 96.3,
                "latency": 38.157,
                "stderr": 0.422,
                "cost_per_test": 0.017279,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 96.2,
                "latency": 8.934,
                "stderr": 0.428,
                "cost_per_test": 0.002862,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 96.15,
                "latency": 23.673,
                "stderr": 0.43,
                "cost_per_test": 0.045694,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 96.1,
                "latency": 11.335,
                "stderr": 0.433,
                "cost_per_test": 0.009828,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "openai/o4-mini-2025-04-16": {
                "accuracy": 96.05,
                "latency": 6.818,
                "stderr": 0.436,
                "cost_per_test": 0.0028,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 96.0,
                "latency": 32.141,
                "stderr": 0.438,
                "cost_per_test": 0.150804,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 95.85,
                "latency": 27.209,
                "stderr": 0.446,
                "cost_per_test": 0.013089,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 95.45,
                "latency": 30.837,
                "stderr": 0.466,
                "cost_per_test": 0.027187,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "alibaba/qwen3.5-plus-thinking": {
                "accuracy": 95.05,
                "latency": 57.556,
                "stderr": 0.485,
                "cost_per_test": 0.008332,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba"
            },
            "openai/o3-mini-2025-01-31": {
                "accuracy": 94.8,
                "latency": 8.393,
                "stderr": 0.976,
                "cost_per_test": 0.005338,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 94.55,
                "latency": 36.104,
                "stderr": 0.508,
                "cost_per_test": 0.031785,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 94.3,
                "latency": 19.327,
                "stderr": 0.518,
                "cost_per_test": 0.009332,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 94.2,
                "latency": 42.58,
                "stderr": 0.523,
                "cost_per_test": 0.007152,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI"
            },
            "zai/glm-4.7": {
                "accuracy": 94.0,
                "latency": 57.619,
                "stderr": 0.531,
                "cost_per_test": 0.005019,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI"
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 93.85,
                "latency": 20.146,
                "stderr": 0.537,
                "cost_per_test": 0.022454,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "zai/glm-5-thinking": {
                "accuracy": 93.85,
                "latency": 81.552,
                "stderr": 0.537,
                "cost_per_test": 0.008061,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI"
            },
            "fireworks/deepseek-v3p2-thinking": {
                "accuracy": 93.75,
                "latency": 54.437,
                "stderr": 0.541,
                "cost_per_test": 0.000783,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 20480,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "openai/gpt-5-nano-2025-08-07": {
                "accuracy": 93.55,
                "latency": 19.043,
                "stderr": 0.549,
                "cost_per_test": 0.001288,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "google/gemini-2.5-pro-exp-03-25": {
                "accuracy": 93.4,
                "latency": 10.641,
                "stderr": 0.555,
                "cost_per_test": 0.002886,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "anthropic/claude-opus-4-5-20251101": {
                "accuracy": 93.0,
                "latency": 10.312,
                "stderr": 0.57,
                "cost_per_test": 0.043243,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "anthropic/claude-opus-4-20250514": {
                "accuracy": 92.9,
                "latency": 11.763,
                "stderr": 0.574,
                "cost_per_test": 0.03822,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "zai/glm-4.6": {
                "accuracy": 92.85,
                "latency": 28.429,
                "stderr": 0.576,
                "cost_per_test": 0.00424,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 122880,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI"
            },
            "anthropic/claude-opus-4-1-20250805-thinking": {
                "accuracy": 92.8,
                "latency": 23.784,
                "stderr": 0.578,
                "cost_per_test": 0.088395,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "minimax/MiniMax-M2.5": {
                "accuracy": 92.6,
                "latency": 33.252,
                "stderr": 0.585,
                "cost_per_test": 0.002917,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax"
            },
            "kimi/kimi-k2-thinking": {
                "accuracy": 92.55,
                "latency": 232.098,
                "stderr": 0.587,
                "cost_per_test": 0.006292,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI"
            },
            "anthropic/claude-opus-4-1-20250805": {
                "accuracy": 92.45,
                "latency": 11.537,
                "stderr": 0.591,
                "cost_per_test": 0.041888,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "anthropic/claude-sonnet-4-20250514-thinking": {
                "accuracy": 92.45,
                "latency": 27.215,
                "stderr": 0.591,
                "cost_per_test": 0.025228,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "grok/grok-2-1212": {
                "accuracy": 91.95,
                "latency": 4.047,
                "stderr": 1.194,
                "cost_per_test": 0.003272,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 0,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "grok/grok-4-fast-reasoning": {
                "accuracy": 91.8,
                "latency": 5.623,
                "stderr": 0.614,
                "cost_per_test": 0.000755,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "grok/grok-4-0709": {
                "accuracy": 91.8,
                "latency": 66.998,
                "stderr": 0.614,
                "cost_per_test": 0.026976,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 91.6,
                "latency": 24.336,
                "stderr": 0.62,
                "cost_per_test": 0.000591,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 91.6,
                "latency": 36.664,
                "stderr": 0.62,
                "cost_per_test": 0.036218,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "fireworks/gpt-oss-120b": {
                "accuracy": 91.4,
                "latency": 25.822,
                "stderr": 0.627,
                "cost_per_test": 0.000602,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "openai/gpt-4.1-2025-04-14": {
                "accuracy": 91.05,
                "latency": 2.959,
                "stderr": 0.638,
                "cost_per_test": 0.002967,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "google/gemini-2.5-flash-preview-04-17-thinking": {
                "accuracy": 91.05,
                "latency": 8.779,
                "stderr": 0.638,
                "cost_per_test": 0.003769,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "google/gemini-2.5-flash-preview-09-2025": {
                "accuracy": 90.85,
                "latency": 6.472,
                "stderr": 0.645,
                "cost_per_test": 0.001264,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "google/gemini-2.5-flash-preview-09-2025-thinking": {
                "accuracy": 90.65,
                "latency": 20.692,
                "stderr": 0.651,
                "cost_per_test": 0.001269,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "fireworks/qwen3-235b-a22b": {
                "accuracy": 90.55,
                "latency": 26.468,
                "stderr": 0.654,
                "cost_per_test": 0.001196,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "fireworks/deepseek-r1": {
                "accuracy": 90.45,
                "latency": 43.736,
                "stderr": 1.289,
                "cost_per_test": 0.005958,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "minimax/MiniMax-M2.1": {
                "accuracy": 90.4,
                "latency": 36.292,
                "stderr": 0.659,
                "cost_per_test": 0.002414,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax"
            },
            "openai/o1-mini-2024-09-12": {
                "accuracy": 90.35,
                "latency": 5.92,
                "stderr": 1.295,
                "cost_per_test": 0.004272,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "anthropic/claude-sonnet-4-20250514": {
                "accuracy": 90.2,
                "latency": 7.448,
                "stderr": 0.665,
                "cost_per_test": 0.007085,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "anthropic/claude-3-7-sonnet-20250219-thinking": {
                "accuracy": 90.1,
                "latency": 15.933,
                "stderr": 1.31,
                "cost_per_test": 0.016584,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 6096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "grok/grok-3-mini-fast-high-reasoning": {
                "accuracy": 89.9,
                "latency": 7.253,
                "stderr": 0.674,
                "cost_per_test": 0.000769,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "mistralai/magistral-medium-2509": {
                "accuracy": 89.9,
                "latency": 17.338,
                "stderr": 0.674,
                "cost_per_test": 0.006903,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI"
            },
            "zai/glm-4.5": {
                "accuracy": 89.9,
                "latency": 74.936,
                "stderr": 0.674,
                "cost_per_test": 0.003416,
                "temperature": 0.6,
                "top_p": 1.0,
                "max_output_tokens": 81920,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI"
            },
            "google/gemini-2.5-flash-lite-preview-09-2025-thinking": {
                "accuracy": 89.1,
                "latency": 12.894,
                "stderr": 0.697,
                "cost_per_test": 0.000167,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "together/meta-llama/Meta-Llama-3.1-405B-Instruct-Turbo": {
                "accuracy": 88.65,
                "latency": 8.614,
                "stderr": 1.391,
                "cost_per_test": 0.002656,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "openai/o1-preview-2024-09-12": {
                "accuracy": 88.65,
                "latency": 14.705,
                "stderr": 1.391,
                "cost_per_test": 0.105072,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "fireworks/deepseek-v3p2": {
                "accuracy": 88.45,
                "latency": 13.222,
                "stderr": 0.715,
                "cost_per_test": 0.000234,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 20480,
                "reasoning": null,
                "reasoning_effort": "none",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "grok/grok-3-mini-fast-low-reasoning": {
                "accuracy": 88.25,
                "latency": 4.855,
                "stderr": 0.72,
                "cost_per_test": 0.000558,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "low",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "openai/gpt-4o-2024-08-06": {
                "accuracy": 88.1,
                "latency": 4.116,
                "stderr": 1.42,
                "cost_per_test": 0.002784,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "alibaba/qwen3-max-preview": {
                "accuracy": 87.5,
                "latency": 17.906,
                "stderr": 0.74,
                "cost_per_test": 0.003191,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba"
            },
            "alibaba/qwen3-max": {
                "accuracy": 86.9,
                "latency": 0.0,
                "stderr": 0.754,
                "cost_per_test": 0.003169,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba"
            },
            "google/gemini-2.5-flash-preview-04-17": {
                "accuracy": 86.6,
                "latency": 2.126,
                "stderr": 0.762,
                "cost_per_test": 0.000993,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "together/meta-llama/Meta-Llama-3.1-70B-Instruct-Turbo": {
                "accuracy": 84.85,
                "latency": 5.149,
                "stderr": 1.571,
                "cost_per_test": 0.000619,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "openai/gpt-4.1-mini-2025-04-14": {
                "accuracy": 84.0,
                "latency": 1.751,
                "stderr": 0.82,
                "cost_per_test": 0.000517,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "together/moonshotai/Kimi-K2-Instruct": {
                "accuracy": 83.9,
                "latency": 12.731,
                "stderr": 0.822,
                "cost_per_test": 0.001034,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "grok/grok-3": {
                "accuracy": 83.7,
                "latency": 8.814,
                "stderr": 0.826,
                "cost_per_test": 0.006281,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "mistralai/magistral-small-2509": {
                "accuracy": 83.15,
                "latency": 7.856,
                "stderr": 0.837,
                "cost_per_test": 0.001716,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI"
            },
            "anthropic/claude-3-5-sonnet-20241022": {
                "accuracy": 82.85,
                "latency": 6.414,
                "stderr": 1.652,
                "cost_per_test": 0.005968,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "fireworks/gpt-oss-20b": {
                "accuracy": 82.7,
                "latency": 30.168,
                "stderr": 0.846,
                "cost_per_test": 0.000338,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "fireworks/deepseek-v3-0324": {
                "accuracy": 82.5,
                "latency": 13.044,
                "stderr": 0.85,
                "cost_per_test": 0.000617,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "mistralai/mistral-large-2512": {
                "accuracy": 82.05,
                "latency": 17.947,
                "stderr": 0.858,
                "cost_per_test": 0.000773,
                "temperature": 0.065,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI"
            },
            "google/gemini-2.0-flash-001": {
                "accuracy": 81.7,
                "latency": 1.928,
                "stderr": 1.694,
                "cost_per_test": 0.000115,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "cohere/command-a-03-2025": {
                "accuracy": 81.05,
                "latency": 2.851,
                "stderr": 0.876,
                "cost_per_test": 0.003452,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 8000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere"
            },
            "google/gemini-2.5-flash-lite-preview-09-2025": {
                "accuracy": 80.65,
                "latency": 5.974,
                "stderr": 0.883,
                "cost_per_test": 0.000278,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "fireworks/deepseek-v3": {
                "accuracy": 80.5,
                "latency": 7.165,
                "stderr": 0.886,
                "cost_per_test": 0.00058,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "openai/gpt-4-turbo": {
                "accuracy": 80.05,
                "latency": 9.057,
                "stderr": 1.751,
                "cost_per_test": 0.012663,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "mistralai/mistral-medium-2505": {
                "accuracy": 78.3,
                "latency": 7.192,
                "stderr": 0.922,
                "cost_per_test": 0.000638,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI"
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 78.2,
                "latency": 24.803,
                "stderr": 0.923,
                "cost_per_test": 0.014581,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "google/gemini-1.5-pro-002": {
                "accuracy": 76.45,
                "latency": 5.804,
                "stderr": 1.858,
                "cost_per_test": 0.002938,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "grok/grok-4-1-fast-non-reasoning": {
                "accuracy": 76.3,
                "latency": 7.339,
                "stderr": 0.951,
                "cost_per_test": 0.000165,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "mistralai/mistral-large-2411": {
                "accuracy": 76.05,
                "latency": 5.074,
                "stderr": 1.869,
                "cost_per_test": 0.002473,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI"
            },
            "grok/grok-4-fast-non-reasoning": {
                "accuracy": 75.05,
                "latency": 2.377,
                "stderr": 0.968,
                "cost_per_test": 0.000482,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "together/Qwen/Qwen2.5-72B-Instruct-Turbo": {
                "accuracy": 74.6,
                "latency": 5.298,
                "stderr": 1.906,
                "cost_per_test": 0.000844,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "openai/gpt-4o-mini-2024-07-18": {
                "accuracy": 72.0,
                "latency": 2.246,
                "stderr": 1.966,
                "cost_per_test": 0.000157,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "mistralai/mistral-small-2503": {
                "accuracy": 68.8,
                "latency": 3.66,
                "stderr": 1.036,
                "cost_per_test": 0.000102,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI"
            },
            "openai/gpt-4.1-nano-2025-04-14": {
                "accuracy": 68.1,
                "latency": 1.409,
                "stderr": 1.042,
                "cost_per_test": 0.000155,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "ai21labs/jamba-1.5-large": {
                "accuracy": 67.65,
                "latency": 7.835,
                "stderr": 2.049,
                "cost_per_test": 0.002199,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs"
            },
            "together/mistralai/Mixtral-8x22B-Instruct-v0.1": {
                "accuracy": 62.5,
                "latency": 6.923,
                "stderr": 2.12,
                "cost_per_test": 0.000917,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "together/meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo": {
                "accuracy": 62.3,
                "latency": 1.878,
                "stderr": 2.122,
                "cost_per_test": 0.000122,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "openai/gpt-3.5-turbo": {
                "accuracy": 58.6,
                "latency": 1.324,
                "stderr": 2.157,
                "cost_per_test": 0.000424,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "mistralai/mistral-small-2402": {
                "accuracy": 56.65,
                "latency": 3.045,
                "stderr": 2.17,
                "cost_per_test": 0.00021,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI"
            },
            "ai21labs/jamba-1.5-mini": {
                "accuracy": 55.3,
                "latency": 1.131,
                "stderr": 2.177,
                "cost_per_test": 0.000148,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs"
            },
            "together/mistralai/Mixtral-8x7B-v0.1": {
                "accuracy": 52.55,
                "latency": 3.0,
                "stderr": 2.186,
                "cost_per_test": 0.000449,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "ai21labs/jamba-mini-1.6": {
                "accuracy": 52.45,
                "latency": 2.265,
                "stderr": 1.117,
                "cost_per_test": 0.000203,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs"
            },
            "ai21labs/jamba-large-1.6": {
                "accuracy": 50.85,
                "latency": 8.096,
                "stderr": 1.118,
                "cost_per_test": 0.003151,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs"
            },
            "together/meta-llama/Llama-4-Scout-17B-16E-Instruct": {
                "accuracy": 49.7,
                "latency": 5.084,
                "stderr": 1.118,
                "cost_per_test": 0.000326,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "fireworks/llama4-maverick-instruct-basic": {
                "accuracy": 43.2,
                "latency": 8.04,
                "stderr": 1.108,
                "cost_per_test": 0.000695,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "cohere/command-r-plus": {
                "accuracy": 2.6,
                "latency": 5.43,
                "stderr": 0.703,
                "cost_per_test": 0.003498,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 4000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere"
            }
        },
        "white": {
            "openai/o1-2024-12-17": {
                "accuracy": 96.65,
                "latency": 11.701,
                "stderr": 0.793,
                "cost_per_test": 0.05222,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 96.45,
                "latency": 11.612,
                "stderr": 0.414,
                "cost_per_test": 0.009964,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 96.35,
                "latency": 37.602,
                "stderr": 0.419,
                "cost_per_test": 0.020483,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 96.3,
                "latency": 27.09,
                "stderr": 0.422,
                "cost_per_test": 0.013138,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 96.25,
                "latency": 37.716,
                "stderr": 0.425,
                "cost_per_test": 0.017949,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 96.2,
                "latency": 8.986,
                "stderr": 0.428,
                "cost_per_test": 0.002866,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 96.2,
                "latency": 31.392,
                "stderr": 0.428,
                "cost_per_test": 0.147985,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 96.15,
                "latency": 21.062,
                "stderr": 0.43,
                "cost_per_test": 0.029315,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 96.05,
                "latency": 23.777,
                "stderr": 0.436,
                "cost_per_test": 0.045844,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "openai/o4-mini-2025-04-16": {
                "accuracy": 96.0,
                "latency": 6.44,
                "stderr": 0.438,
                "cost_per_test": 0.002789,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "openai/o3-2025-04-16": {
                "accuracy": 95.95,
                "latency": 8.655,
                "stderr": 0.441,
                "cost_per_test": 0.005413,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 95.75,
                "latency": 30.029,
                "stderr": 0.451,
                "cost_per_test": 0.026998,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "alibaba/qwen3.5-plus-thinking": {
                "accuracy": 95.2,
                "latency": 56.656,
                "stderr": 0.478,
                "cost_per_test": 0.008219,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba"
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 94.85,
                "latency": 36.589,
                "stderr": 0.494,
                "cost_per_test": 0.031318,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "openai/o3-mini-2025-01-31": {
                "accuracy": 94.65,
                "latency": 8.21,
                "stderr": 0.989,
                "cost_per_test": 0.005385,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 94.5,
                "latency": 8.442,
                "stderr": 0.51,
                "cost_per_test": 0.008982,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "zai/glm-5-thinking": {
                "accuracy": 94.4,
                "latency": 80.405,
                "stderr": 0.514,
                "cost_per_test": 0.007949,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI"
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 94.3,
                "latency": 41.102,
                "stderr": 0.518,
                "cost_per_test": 0.006913,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI"
            },
            "fireworks/deepseek-v3p2-thinking": {
                "accuracy": 94.1,
                "latency": 52.717,
                "stderr": 0.527,
                "cost_per_test": 0.000759,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 20480,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "zai/glm-4.7": {
                "accuracy": 94.1,
                "latency": 54.962,
                "stderr": 0.527,
                "cost_per_test": 0.004822,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI"
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 94.05,
                "latency": 19.281,
                "stderr": 0.529,
                "cost_per_test": 0.021316,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "anthropic/claude-opus-4-1-20250805-thinking": {
                "accuracy": 93.95,
                "latency": 23.659,
                "stderr": 0.533,
                "cost_per_test": 0.088511,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "openai/gpt-5-nano-2025-08-07": {
                "accuracy": 93.65,
                "latency": 27.294,
                "stderr": 0.545,
                "cost_per_test": 0.00118,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "openai/o1-preview-2024-09-12": {
                "accuracy": 93.55,
                "latency": 17.939,
                "stderr": 1.079,
                "cost_per_test": 0.110351,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "google/gemini-2.5-pro-exp-03-25": {
                "accuracy": 93.3,
                "latency": 10.552,
                "stderr": 0.559,
                "cost_per_test": 0.002878,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "anthropic/claude-opus-4-20250514": {
                "accuracy": 93.3,
                "latency": 11.9,
                "stderr": 0.559,
                "cost_per_test": 0.03803,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "grok/grok-4-0709": {
                "accuracy": 93.2,
                "latency": 62.836,
                "stderr": 0.563,
                "cost_per_test": 0.024839,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "anthropic/claude-opus-4-5-20251101": {
                "accuracy": 93.15,
                "latency": 10.322,
                "stderr": 0.565,
                "cost_per_test": 0.043257,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "anthropic/claude-sonnet-4-20250514-thinking": {
                "accuracy": 93.05,
                "latency": 27.551,
                "stderr": 0.569,
                "cost_per_test": 0.025419,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "kimi/kimi-k2-thinking": {
                "accuracy": 92.85,
                "latency": 221.119,
                "stderr": 0.576,
                "cost_per_test": 0.006098,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI"
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 92.55,
                "latency": 23.339,
                "stderr": 0.587,
                "cost_per_test": 0.000552,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "grok/grok-2-1212": {
                "accuracy": 92.5,
                "latency": 4.005,
                "stderr": 1.156,
                "cost_per_test": 0.003244,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 0,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "anthropic/claude-opus-4-1-20250805": {
                "accuracy": 92.5,
                "latency": 25.055,
                "stderr": 0.589,
                "cost_per_test": 0.041493,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "minimax/MiniMax-M2.5": {
                "accuracy": 92.35,
                "latency": 33.532,
                "stderr": 0.594,
                "cost_per_test": 0.002954,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax"
            },
            "google/gemini-2.5-flash-preview-09-2025": {
                "accuracy": 92.3,
                "latency": 6.538,
                "stderr": 0.596,
                "cost_per_test": 0.001279,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "zai/glm-4.6": {
                "accuracy": 92.2,
                "latency": 28.507,
                "stderr": 0.6,
                "cost_per_test": 0.004307,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 122880,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI"
            },
            "grok/grok-4-fast-reasoning": {
                "accuracy": 92.15,
                "latency": 5.22,
                "stderr": 0.601,
                "cost_per_test": 0.000735,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 92.0,
                "latency": 34.904,
                "stderr": 0.607,
                "cost_per_test": 0.035394,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "google/gemini-2.5-flash-preview-04-17-thinking": {
                "accuracy": 91.9,
                "latency": 8.595,
                "stderr": 0.61,
                "cost_per_test": 0.003689,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "openai/gpt-4.1-2025-04-14": {
                "accuracy": 91.55,
                "latency": 3.289,
                "stderr": 0.622,
                "cost_per_test": 0.002961,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "google/gemini-2.5-flash-preview-09-2025-thinking": {
                "accuracy": 91.4,
                "latency": 10.196,
                "stderr": 0.627,
                "cost_per_test": 0.001265,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "fireworks/gpt-oss-120b": {
                "accuracy": 91.3,
                "latency": 4.513,
                "stderr": 0.63,
                "cost_per_test": 0.000611,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "fireworks/deepseek-r1": {
                "accuracy": 90.95,
                "latency": 38.107,
                "stderr": 1.259,
                "cost_per_test": 0.005672,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "anthropic/claude-sonnet-4-20250514": {
                "accuracy": 90.85,
                "latency": 7.627,
                "stderr": 0.645,
                "cost_per_test": 0.007094,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "fireworks/qwen3-235b-a22b": {
                "accuracy": 90.85,
                "latency": 25.266,
                "stderr": 0.645,
                "cost_per_test": 0.001157,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "minimax/MiniMax-M2.1": {
                "accuracy": 90.8,
                "latency": 34.686,
                "stderr": 0.646,
                "cost_per_test": 0.00232,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax"
            },
            "grok/grok-3-mini-fast-high-reasoning": {
                "accuracy": 90.3,
                "latency": 6.958,
                "stderr": 0.662,
                "cost_per_test": 0.000761,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "openai/o1-mini-2024-09-12": {
                "accuracy": 90.1,
                "latency": 5.869,
                "stderr": 1.31,
                "cost_per_test": 0.004311,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "anthropic/claude-3-7-sonnet-20250219-thinking": {
                "accuracy": 89.95,
                "latency": 15.512,
                "stderr": 1.319,
                "cost_per_test": 0.016648,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 6096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "zai/glm-4.5": {
                "accuracy": 89.85,
                "latency": 61.729,
                "stderr": 0.675,
                "cost_per_test": 0.003485,
                "temperature": 0.6,
                "top_p": 1.0,
                "max_output_tokens": 81920,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI"
            },
            "fireworks/deepseek-v3p2": {
                "accuracy": 89.35,
                "latency": 13.182,
                "stderr": 0.69,
                "cost_per_test": 0.000233,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 20480,
                "reasoning": null,
                "reasoning_effort": "none",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "grok/grok-3-mini-fast-low-reasoning": {
                "accuracy": 89.2,
                "latency": 4.796,
                "stderr": 0.694,
                "cost_per_test": 0.000558,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "low",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "google/gemini-2.5-flash-lite-preview-09-2025-thinking": {
                "accuracy": 89.05,
                "latency": 13.816,
                "stderr": 0.698,
                "cost_per_test": 0.00017,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "together/meta-llama/Meta-Llama-3.1-405B-Instruct-Turbo": {
                "accuracy": 88.95,
                "latency": 11.362,
                "stderr": 1.375,
                "cost_per_test": 0.002655,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "openai/gpt-4o-2024-08-06": {
                "accuracy": 88.5,
                "latency": 3.231,
                "stderr": 1.399,
                "cost_per_test": 0.002794,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "mistralai/magistral-medium-2509": {
                "accuracy": 88.5,
                "latency": 17.086,
                "stderr": 0.713,
                "cost_per_test": 0.006891,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI"
            },
            "alibaba/qwen3-max": {
                "accuracy": 87.45,
                "latency": 0.0,
                "stderr": 0.741,
                "cost_per_test": 0.003181,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba"
            },
            "alibaba/qwen3-max-preview": {
                "accuracy": 87.25,
                "latency": 21.056,
                "stderr": 0.746,
                "cost_per_test": 0.003167,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba"
            },
            "google/gemini-2.5-flash-preview-04-17": {
                "accuracy": 87.0,
                "latency": 1.954,
                "stderr": 0.752,
                "cost_per_test": 0.000891,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "together/meta-llama/Meta-Llama-3.1-70B-Instruct-Turbo": {
                "accuracy": 84.75,
                "latency": 4.534,
                "stderr": 1.576,
                "cost_per_test": 0.00062,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "openai/gpt-4.1-mini-2025-04-14": {
                "accuracy": 84.7,
                "latency": 1.768,
                "stderr": 0.805,
                "cost_per_test": 0.000518,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "together/moonshotai/Kimi-K2-Instruct": {
                "accuracy": 84.05,
                "latency": 12.55,
                "stderr": 0.819,
                "cost_per_test": 0.001025,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "grok/grok-3": {
                "accuracy": 83.8,
                "latency": 5.838,
                "stderr": 0.824,
                "cost_per_test": 0.006131,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "anthropic/claude-3-5-sonnet-20241022": {
                "accuracy": 83.5,
                "latency": 6.112,
                "stderr": 1.626,
                "cost_per_test": 0.005951,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "mistralai/mistral-large-2512": {
                "accuracy": 82.8,
                "latency": 17.683,
                "stderr": 0.844,
                "cost_per_test": 0.000771,
                "temperature": 0.065,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI"
            },
            "google/gemini-2.0-flash-001": {
                "accuracy": 82.2,
                "latency": 1.928,
                "stderr": 1.676,
                "cost_per_test": 0.000113,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "mistralai/magistral-small-2509": {
                "accuracy": 82.1,
                "latency": 8.073,
                "stderr": 0.857,
                "cost_per_test": 0.001731,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI"
            },
            "fireworks/deepseek-v3-0324": {
                "accuracy": 81.85,
                "latency": 13.031,
                "stderr": 0.862,
                "cost_per_test": 0.000617,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "fireworks/gpt-oss-20b": {
                "accuracy": 81.7,
                "latency": 26.591,
                "stderr": 0.865,
                "cost_per_test": 0.000332,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "fireworks/deepseek-v3": {
                "accuracy": 81.35,
                "latency": 6.692,
                "stderr": 0.871,
                "cost_per_test": 0.000579,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "openai/gpt-4-turbo": {
                "accuracy": 81.1,
                "latency": 9.443,
                "stderr": 1.715,
                "cost_per_test": 0.012734,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "cohere/command-a-03-2025": {
                "accuracy": 80.65,
                "latency": 2.947,
                "stderr": 0.883,
                "cost_per_test": 0.00347,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 8000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere"
            },
            "google/gemini-2.5-flash-lite-preview-09-2025": {
                "accuracy": 80.6,
                "latency": 6.004,
                "stderr": 0.884,
                "cost_per_test": 0.000285,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 79.4,
                "latency": 23.618,
                "stderr": 0.904,
                "cost_per_test": 0.013893,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "mistralai/mistral-medium-2505": {
                "accuracy": 78.4,
                "latency": 7.442,
                "stderr": 0.92,
                "cost_per_test": 0.00065,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI"
            },
            "google/gemini-1.5-pro-002": {
                "accuracy": 76.75,
                "latency": 6.077,
                "stderr": 1.85,
                "cost_per_test": 0.002945,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "mistralai/mistral-large-2411": {
                "accuracy": 76.45,
                "latency": 4.752,
                "stderr": 1.858,
                "cost_per_test": 0.002411,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI"
            },
            "grok/grok-4-1-fast-non-reasoning": {
                "accuracy": 76.0,
                "latency": 7.348,
                "stderr": 0.955,
                "cost_per_test": 0.000166,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "grok/grok-4-fast-non-reasoning": {
                "accuracy": 75.4,
                "latency": 2.555,
                "stderr": 0.963,
                "cost_per_test": 0.000484,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "together/Qwen/Qwen2.5-72B-Instruct-Turbo": {
                "accuracy": 74.2,
                "latency": 5.909,
                "stderr": 1.916,
                "cost_per_test": 0.000851,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "openai/gpt-4o-mini-2024-07-18": {
                "accuracy": 72.35,
                "latency": 1.776,
                "stderr": 1.959,
                "cost_per_test": 0.000157,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "mistralai/mistral-small-2503": {
                "accuracy": 69.95,
                "latency": 3.662,
                "stderr": 1.025,
                "cost_per_test": 0.000102,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI"
            },
            "openai/gpt-4.1-nano-2025-04-14": {
                "accuracy": 68.9,
                "latency": 1.51,
                "stderr": 1.035,
                "cost_per_test": 0.000154,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "ai21labs/jamba-1.5-large": {
                "accuracy": 68.45,
                "latency": 8.122,
                "stderr": 2.035,
                "cost_per_test": 0.002214,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs"
            },
            "together/meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo": {
                "accuracy": 63.45,
                "latency": 1.836,
                "stderr": 2.109,
                "cost_per_test": 0.000122,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "together/mistralai/Mixtral-8x22B-Instruct-v0.1": {
                "accuracy": 62.15,
                "latency": 10.952,
                "stderr": 2.124,
                "cost_per_test": 0.000923,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "openai/gpt-3.5-turbo": {
                "accuracy": 58.1,
                "latency": 2.002,
                "stderr": 2.16,
                "cost_per_test": 0.000426,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "mistralai/mistral-small-2402": {
                "accuracy": 57.25,
                "latency": 2.547,
                "stderr": 2.166,
                "cost_per_test": 0.00021,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI"
            },
            "ai21labs/jamba-1.5-mini": {
                "accuracy": 55.9,
                "latency": 1.117,
                "stderr": 2.174,
                "cost_per_test": 0.000147,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs"
            },
            "together/meta-llama/Llama-4-Scout-17B-16E-Instruct": {
                "accuracy": 53.95,
                "latency": 4.926,
                "stderr": 1.114,
                "cost_per_test": 0.000324,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "ai21labs/jamba-mini-1.6": {
                "accuracy": 53.85,
                "latency": 2.123,
                "stderr": 1.115,
                "cost_per_test": 0.000202,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs"
            },
            "together/mistralai/Mixtral-8x7B-v0.1": {
                "accuracy": 53.25,
                "latency": 2.669,
                "stderr": 2.185,
                "cost_per_test": 0.000451,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "ai21labs/jamba-large-1.6": {
                "accuracy": 51.3,
                "latency": 6.955,
                "stderr": 1.118,
                "cost_per_test": 0.003163,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs"
            },
            "fireworks/llama4-maverick-instruct-basic": {
                "accuracy": 44.0,
                "latency": 7.626,
                "stderr": 1.11,
                "cost_per_test": 0.00071,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "cohere/command-r-plus": {
                "accuracy": 2.55,
                "latency": 6.126,
                "stderr": 0.696,
                "cost_per_test": 0.00346,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 4000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere"
            }
        },
        "indigenous": {
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 96.3,
                "latency": 11.949,
                "stderr": 0.422,
                "cost_per_test": 0.010159,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "openai/o1-2024-12-17": {
                "accuracy": 96.15,
                "latency": 11.738,
                "stderr": 0.847,
                "cost_per_test": 0.054309,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 96.1,
                "latency": 39.948,
                "stderr": 0.433,
                "cost_per_test": 0.018043,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 96.05,
                "latency": 38.864,
                "stderr": 0.436,
                "cost_per_test": 0.021321,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "openai/o4-mini-2025-04-16": {
                "accuracy": 95.9,
                "latency": 6.572,
                "stderr": 0.443,
                "cost_per_test": 0.00285,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "openai/o3-2025-04-16": {
                "accuracy": 95.9,
                "latency": 8.923,
                "stderr": 0.443,
                "cost_per_test": 0.005537,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 95.9,
                "latency": 21.919,
                "stderr": 0.443,
                "cost_per_test": 0.029866,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 95.9,
                "latency": 24.032,
                "stderr": 0.443,
                "cost_per_test": 0.045777,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 95.6,
                "latency": 26.722,
                "stderr": 0.459,
                "cost_per_test": 0.012851,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 95.55,
                "latency": 32.345,
                "stderr": 0.461,
                "cost_per_test": 0.150479,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 95.4,
                "latency": 23.367,
                "stderr": 0.468,
                "cost_per_test": 0.003247,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 95.2,
                "latency": 30.446,
                "stderr": 0.478,
                "cost_per_test": 0.026669,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 94.75,
                "latency": 36.056,
                "stderr": 0.499,
                "cost_per_test": 0.031626,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 94.65,
                "latency": 8.736,
                "stderr": 0.503,
                "cost_per_test": 0.009334,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "openai/o3-mini-2025-01-31": {
                "accuracy": 94.65,
                "latency": 8.855,
                "stderr": 0.989,
                "cost_per_test": 0.005528,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 94.55,
                "latency": 42.361,
                "stderr": 0.508,
                "cost_per_test": 0.007132,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI"
            },
            "alibaba/qwen3.5-plus-thinking": {
                "accuracy": 94.5,
                "latency": 45.854,
                "stderr": 0.51,
                "cost_per_test": 0.008759,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba"
            },
            "zai/glm-5-thinking": {
                "accuracy": 94.3,
                "latency": 82.724,
                "stderr": 0.518,
                "cost_per_test": 0.008069,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI"
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 93.95,
                "latency": 20.794,
                "stderr": 0.533,
                "cost_per_test": 0.022737,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "openai/o1-preview-2024-09-12": {
                "accuracy": 93.5,
                "latency": 18.077,
                "stderr": 1.083,
                "cost_per_test": 0.110786,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "fireworks/deepseek-v3p2-thinking": {
                "accuracy": 93.45,
                "latency": 54.575,
                "stderr": 0.553,
                "cost_per_test": 0.000785,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 20480,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "anthropic/claude-opus-4-1-20250805-thinking": {
                "accuracy": 93.35,
                "latency": 23.579,
                "stderr": 0.557,
                "cost_per_test": 0.088503,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "google/gemini-2.5-pro-exp-03-25": {
                "accuracy": 93.2,
                "latency": 11.072,
                "stderr": 0.563,
                "cost_per_test": 0.00294,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "anthropic/claude-opus-4-5-20251101": {
                "accuracy": 92.85,
                "latency": 10.511,
                "stderr": 0.576,
                "cost_per_test": 0.043491,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "openai/gpt-5-nano-2025-08-07": {
                "accuracy": 92.8,
                "latency": 15.634,
                "stderr": 0.578,
                "cost_per_test": 0.001345,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "anthropic/claude-opus-4-20250514": {
                "accuracy": 92.7,
                "latency": 12.307,
                "stderr": 0.582,
                "cost_per_test": 0.038289,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "anthropic/claude-sonnet-4-20250514-thinking": {
                "accuracy": 92.6,
                "latency": 26.42,
                "stderr": 0.585,
                "cost_per_test": 0.025093,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "minimax/MiniMax-M2.5": {
                "accuracy": 92.6,
                "latency": 34.367,
                "stderr": 0.585,
                "cost_per_test": 0.003003,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax"
            },
            "zai/glm-4.7": {
                "accuracy": 92.55,
                "latency": 50.876,
                "stderr": 0.587,
                "cost_per_test": 0.00497,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI"
            },
            "grok/grok-4-0709": {
                "accuracy": 92.4,
                "latency": 63.94,
                "stderr": 0.593,
                "cost_per_test": 0.025788,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "zai/glm-4.6": {
                "accuracy": 92.3,
                "latency": 29.075,
                "stderr": 0.596,
                "cost_per_test": 0.004167,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 122880,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI"
            },
            "anthropic/claude-opus-4-1-20250805": {
                "accuracy": 92.05,
                "latency": 12.513,
                "stderr": 0.605,
                "cost_per_test": 0.041815,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "grok/grok-4-fast-reasoning": {
                "accuracy": 92.0,
                "latency": 5.407,
                "stderr": 0.607,
                "cost_per_test": 0.000752,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 91.9,
                "latency": 36.533,
                "stderr": 0.61,
                "cost_per_test": 0.036776,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "google/gemini-2.5-flash-preview-09-2025": {
                "accuracy": 91.85,
                "latency": 6.525,
                "stderr": 0.612,
                "cost_per_test": 0.001277,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "kimi/kimi-k2-thinking": {
                "accuracy": 91.85,
                "latency": 226.732,
                "stderr": 0.612,
                "cost_per_test": 0.006252,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI"
            },
            "grok/grok-2-1212": {
                "accuracy": 91.6,
                "latency": 4.288,
                "stderr": 1.217,
                "cost_per_test": 0.003412,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 0,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "fireworks/gpt-oss-120b": {
                "accuracy": 91.6,
                "latency": 29.72,
                "stderr": 0.62,
                "cost_per_test": 0.00061,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 91.55,
                "latency": 14.818,
                "stderr": 0.622,
                "cost_per_test": 0.000582,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "minimax/MiniMax-M2.1": {
                "accuracy": 91.25,
                "latency": 35.943,
                "stderr": 0.632,
                "cost_per_test": 0.002371,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax"
            },
            "google/gemini-2.5-flash-preview-09-2025-thinking": {
                "accuracy": 91.05,
                "latency": 28.58,
                "stderr": 0.638,
                "cost_per_test": 0.001266,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "openai/gpt-4.1-2025-04-14": {
                "accuracy": 90.9,
                "latency": 2.968,
                "stderr": 0.643,
                "cost_per_test": 0.002986,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "fireworks/deepseek-r1": {
                "accuracy": 90.1,
                "latency": 41.724,
                "stderr": 1.31,
                "cost_per_test": 0.006038,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "google/gemini-2.5-flash-preview-04-17-thinking": {
                "accuracy": 90.0,
                "latency": 9.267,
                "stderr": 0.671,
                "cost_per_test": 0.004007,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "fireworks/qwen3-235b-a22b": {
                "accuracy": 89.9,
                "latency": 26.484,
                "stderr": 0.674,
                "cost_per_test": 0.001205,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "openai/o1-mini-2024-09-12": {
                "accuracy": 89.8,
                "latency": 5.939,
                "stderr": 1.327,
                "cost_per_test": 0.004375,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "anthropic/claude-sonnet-4-20250514": {
                "accuracy": 89.65,
                "latency": 7.539,
                "stderr": 0.681,
                "cost_per_test": 0.007085,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "fireworks/deepseek-v3p2": {
                "accuracy": 89.65,
                "latency": 13.335,
                "stderr": 0.681,
                "cost_per_test": 0.000237,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 20480,
                "reasoning": null,
                "reasoning_effort": "none",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "anthropic/claude-3-7-sonnet-20250219-thinking": {
                "accuracy": 89.65,
                "latency": 15.994,
                "stderr": 1.336,
                "cost_per_test": 0.016849,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 6096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "grok/grok-3-mini-fast-high-reasoning": {
                "accuracy": 89.25,
                "latency": 7.067,
                "stderr": 0.693,
                "cost_per_test": 0.000773,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "mistralai/magistral-medium-2509": {
                "accuracy": 89.1,
                "latency": 17.268,
                "stderr": 0.697,
                "cost_per_test": 0.006966,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI"
            },
            "zai/glm-4.5": {
                "accuracy": 88.95,
                "latency": 64.375,
                "stderr": 0.701,
                "cost_per_test": 0.003343,
                "temperature": 0.6,
                "top_p": 1.0,
                "max_output_tokens": 81920,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI"
            },
            "together/meta-llama/Meta-Llama-3.1-405B-Instruct-Turbo": {
                "accuracy": 88.6,
                "latency": 11.756,
                "stderr": 1.394,
                "cost_per_test": 0.002695,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "google/gemini-2.5-flash-lite-preview-09-2025-thinking": {
                "accuracy": 88.2,
                "latency": 6.932,
                "stderr": 0.721,
                "cost_per_test": 0.000168,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "grok/grok-3-mini-fast-low-reasoning": {
                "accuracy": 87.7,
                "latency": 5.04,
                "stderr": 0.734,
                "cost_per_test": 0.000564,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "low",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "openai/gpt-4o-2024-08-06": {
                "accuracy": 87.65,
                "latency": 3.208,
                "stderr": 1.442,
                "cost_per_test": 0.002805,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "alibaba/qwen3-max-preview": {
                "accuracy": 87.25,
                "latency": 21.108,
                "stderr": 0.746,
                "cost_per_test": 0.003286,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba"
            },
            "alibaba/qwen3-max": {
                "accuracy": 87.0,
                "latency": 0.0,
                "stderr": 0.752,
                "cost_per_test": 0.003224,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba"
            },
            "google/gemini-2.5-flash-preview-04-17": {
                "accuracy": 86.25,
                "latency": 2.359,
                "stderr": 0.77,
                "cost_per_test": 0.001092,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "openai/gpt-4.1-mini-2025-04-14": {
                "accuracy": 84.3,
                "latency": 2.06,
                "stderr": 0.814,
                "cost_per_test": 0.000541,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "together/meta-llama/Meta-Llama-3.1-70B-Instruct-Turbo": {
                "accuracy": 83.4,
                "latency": 7.352,
                "stderr": 1.63,
                "cost_per_test": 0.000625,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "together/moonshotai/Kimi-K2-Instruct": {
                "accuracy": 83.1,
                "latency": 7.657,
                "stderr": 0.838,
                "cost_per_test": 0.001035,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "grok/grok-3": {
                "accuracy": 82.75,
                "latency": 10.571,
                "stderr": 0.845,
                "cost_per_test": 0.006417,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "anthropic/claude-3-5-sonnet-20241022": {
                "accuracy": 82.45,
                "latency": 5.684,
                "stderr": 1.667,
                "cost_per_test": 0.006018,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "fireworks/gpt-oss-20b": {
                "accuracy": 82.4,
                "latency": 4.499,
                "stderr": 0.852,
                "cost_per_test": 0.00034,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "mistralai/magistral-small-2509": {
                "accuracy": 82.05,
                "latency": 8.132,
                "stderr": 0.858,
                "cost_per_test": 0.001732,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI"
            },
            "mistralai/mistral-large-2512": {
                "accuracy": 81.75,
                "latency": 17.583,
                "stderr": 0.864,
                "cost_per_test": 0.000807,
                "temperature": 0.065,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI"
            },
            "fireworks/deepseek-v3-0324": {
                "accuracy": 81.35,
                "latency": 13.213,
                "stderr": 0.871,
                "cost_per_test": 0.00062,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "openai/gpt-4-turbo": {
                "accuracy": 81.05,
                "latency": 9.756,
                "stderr": 1.717,
                "cost_per_test": 0.012809,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "google/gemini-2.0-flash-001": {
                "accuracy": 80.8,
                "latency": 2.007,
                "stderr": 1.726,
                "cost_per_test": 0.000116,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "google/gemini-2.5-flash-lite-preview-09-2025": {
                "accuracy": 79.9,
                "latency": 2.107,
                "stderr": 0.896,
                "cost_per_test": 0.000281,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "fireworks/deepseek-v3": {
                "accuracy": 79.8,
                "latency": 6.096,
                "stderr": 0.898,
                "cost_per_test": 0.000583,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "cohere/command-a-03-2025": {
                "accuracy": 79.4,
                "latency": 2.958,
                "stderr": 0.904,
                "cost_per_test": 0.003488,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 8000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere"
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 78.85,
                "latency": 24.02,
                "stderr": 0.913,
                "cost_per_test": 0.014155,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic"
            },
            "mistralai/mistral-medium-2505": {
                "accuracy": 78.1,
                "latency": 7.408,
                "stderr": 0.925,
                "cost_per_test": 0.000639,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI"
            },
            "mistralai/mistral-large-2411": {
                "accuracy": 76.1,
                "latency": 4.944,
                "stderr": 1.868,
                "cost_per_test": 0.002476,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI"
            },
            "google/gemini-1.5-pro-002": {
                "accuracy": 76.1,
                "latency": 6.143,
                "stderr": 1.868,
                "cost_per_test": 0.002998,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google"
            },
            "grok/grok-4-1-fast-non-reasoning": {
                "accuracy": 75.35,
                "latency": 7.393,
                "stderr": 0.964,
                "cost_per_test": 0.000167,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "grok/grok-4-fast-non-reasoning": {
                "accuracy": 74.6,
                "latency": 2.483,
                "stderr": 0.973,
                "cost_per_test": 0.000485,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI"
            },
            "together/Qwen/Qwen2.5-72B-Instruct-Turbo": {
                "accuracy": 74.05,
                "latency": 6.391,
                "stderr": 1.92,
                "cost_per_test": 0.000852,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "openai/gpt-4o-mini-2024-07-18": {
                "accuracy": 72.2,
                "latency": 2.58,
                "stderr": 1.962,
                "cost_per_test": 0.000158,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "mistralai/mistral-small-2503": {
                "accuracy": 69.35,
                "latency": 3.76,
                "stderr": 1.031,
                "cost_per_test": 0.000104,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI"
            },
            "openai/gpt-4.1-nano-2025-04-14": {
                "accuracy": 67.55,
                "latency": 1.453,
                "stderr": 1.047,
                "cost_per_test": 0.000155,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "ai21labs/jamba-1.5-large": {
                "accuracy": 67.15,
                "latency": 3.992,
                "stderr": 2.057,
                "cost_per_test": 0.002192,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs"
            },
            "together/meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo": {
                "accuracy": 62.55,
                "latency": 1.767,
                "stderr": 2.119,
                "cost_per_test": 0.000122,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "together/mistralai/Mixtral-8x22B-Instruct-v0.1": {
                "accuracy": 61.7,
                "latency": 11.466,
                "stderr": 2.128,
                "cost_per_test": 0.000928,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "openai/gpt-3.5-turbo": {
                "accuracy": 58.15,
                "latency": 1.356,
                "stderr": 2.16,
                "cost_per_test": 0.000428,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI"
            },
            "mistralai/mistral-small-2402": {
                "accuracy": 56.85,
                "latency": 2.713,
                "stderr": 2.169,
                "cost_per_test": 0.000211,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI"
            },
            "ai21labs/jamba-1.5-mini": {
                "accuracy": 54.0,
                "latency": 1.146,
                "stderr": 2.182,
                "cost_per_test": 0.000149,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs"
            },
            "together/mistralai/Mixtral-8x7B-v0.1": {
                "accuracy": 53.8,
                "latency": 2.95,
                "stderr": 2.183,
                "cost_per_test": 0.000457,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "ai21labs/jamba-mini-1.6": {
                "accuracy": 52.8,
                "latency": 2.183,
                "stderr": 1.116,
                "cost_per_test": 0.000204,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs"
            },
            "ai21labs/jamba-large-1.6": {
                "accuracy": 50.3,
                "latency": 7.565,
                "stderr": 1.118,
                "cost_per_test": 0.003166,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs"
            },
            "together/meta-llama/Llama-4-Scout-17B-16E-Instruct": {
                "accuracy": 48.25,
                "latency": 4.617,
                "stderr": 1.117,
                "cost_per_test": 0.000329,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI"
            },
            "fireworks/llama4-maverick-instruct-basic": {
                "accuracy": 39.45,
                "latency": 7.688,
                "stderr": 1.093,
                "cost_per_test": 0.000685,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI"
            },
            "cohere/command-r-plus": {
                "accuracy": 3.05,
                "latency": 5.422,
                "stderr": 0.758,
                "cost_per_test": 0.003452,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 4000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere"
            }
        }
    }
}