{
    "metadata": {
        "benchmark": "CorpFin v2",
        "slug": "corp_fin_v2",
        "description": "A private benchmark evaluating understanding of long-context credit agreements",
        "benchmark_id": "corp_fin_v2",
        "family": "corp_fin",
        "version": "2",
        "updated": "2026-08-12",
        "dataset_type": "private",
        "industry": "finance",
        "tasks": {
            "overall": "Overall",
            "exact_pages": "Exact Pages",
            "max_fitting_context": "Max Fitting Context",
            "shared_max_context": "Shared Max Context"
        },
        "models": [
            "ai21labs/jamba-1.5-large",
            "ai21labs/jamba-1.5-mini",
            "ai21labs/jamba-large-1.6",
            "ai21labs/jamba-mini-1.6",
            "alibaba/qwen3-max",
            "alibaba/qwen3-max-2026-01-23",
            "alibaba/qwen3-max-preview",
            "alibaba/qwen3.5-flash",
            "alibaba/qwen3.5-plus-thinking",
            "alibaba/qwen3.6-27b",
            "alibaba/qwen3.6-max-preview",
            "alibaba/qwen3.6-plus",
            "alibaba/qwen3.7-max",
            "alibaba/qwen3.8-max",
            "ant/ling-3.0-flash-2607",
            "anthropic/claude-3-5-haiku-20241022",
            "anthropic/claude-3-5-sonnet-20241022",
            "anthropic/claude-3-7-sonnet-20250219-thinking",
            "anthropic/claude-fable-5",
            "anthropic/claude-haiku-4-5-20251001",
            "anthropic/claude-haiku-4-5-20251001-thinking",
            "anthropic/claude-opus-4-5-20251101",
            "anthropic/claude-opus-4-5-20251101-thinking",
            "anthropic/claude-opus-4-6-thinking",
            "anthropic/claude-opus-4-7",
            "anthropic/claude-opus-4-8",
            "anthropic/claude-opus-5",
            "anthropic/claude-sonnet-4-20250514",
            "anthropic/claude-sonnet-4-20250514-thinking",
            "anthropic/claude-sonnet-4-5-20250929",
            "anthropic/claude-sonnet-4-5-20250929-thinking",
            "anthropic/claude-sonnet-4-6",
            "anthropic/claude-sonnet-5",
            "arcee-ai/trinity-large-thinking",
            "cohere/command-a-03-2025",
            "deepseek/deepseek-v4-flash-0731",
            "deepseek/deepseek-v4-pro",
            "deepseek/deepseek-v4-pro-0813",
            "fireworks/deepseek-r1",
            "fireworks/deepseek-v3",
            "fireworks/deepseek-v3-0324",
            "fireworks/deepseek-v3p1",
            "fireworks/deepseek-v3p2",
            "fireworks/deepseek-v3p2-thinking",
            "fireworks/gpt-oss-120b",
            "fireworks/gpt-oss-20b",
            "fireworks/llama4-maverick-instruct-basic",
            "fireworks/nemotron-lightning-3p5-30b-a3b",
            "google/gemini-1.5-flash-001",
            "google/gemini-1.5-flash-002",
            "google/gemini-1.5-pro-002",
            "google/gemini-2.0-flash-001",
            "google/gemini-2.0-pro-exp-02-05",
            "google/gemini-2.5-flash-lite-preview-09-2025",
            "google/gemini-2.5-flash-lite-preview-09-2025-thinking",
            "google/gemini-2.5-flash-preview-04-17",
            "google/gemini-2.5-flash-preview-09-2025",
            "google/gemini-2.5-flash-preview-09-2025-thinking",
            "google/gemini-2.5-pro",
            "google/gemini-2.5-pro-exp-03-25",
            "google/gemini-3-flash-preview",
            "google/gemini-3-pro-preview",
            "google/gemini-3.1-flash-lite-preview",
            "google/gemini-3.1-pro-preview",
            "google/gemini-3.5-flash",
            "google/gemini-3.5-flash-lite",
            "google/gemini-3.6-flash",
            "grok/grok-2-1212",
            "grok/grok-3",
            "grok/grok-3-mini-fast-high-reasoning",
            "grok/grok-3-mini-fast-low-reasoning",
            "grok/grok-4-0709",
            "grok/grok-4-1-fast-non-reasoning",
            "grok/grok-4-1-fast-reasoning",
            "grok/grok-4-fast-non-reasoning",
            "grok/grok-4-fast-reasoning",
            "grok/grok-4.20-0309-reasoning",
            "grok/grok-4.3",
            "grok/grok-4.5",
            "grok/grok-4.6",
            "kimi/kimi-k2-thinking",
            "kimi/kimi-k2.5-thinking",
            "kimi/kimi-k2.6",
            "kimi/kimi-k3",
            "meta/muse_spark",
            "meta/muse_spark_1_1",
            "meta/muse_spark_1_2",
            "minimax/MiniMax-M2.1",
            "minimax/MiniMax-M2.5",
            "minimax/MiniMax-M2.7",
            "minimax/MiniMax-M3",
            "mistralai/magistral-medium-2509",
            "mistralai/magistral-small-2509",
            "mistralai/mistral-large-2512",
            "mistralai/mistral-medium-2505",
            "mistralai/mistral-medium-3.5",
            "mistralai/mistral-small-2503",
            "nvidia/nemotron-3-ultra-550b-a55b",
            "openai/gpt-4.1-2025-04-14",
            "openai/gpt-4.1-mini-2025-04-14",
            "openai/gpt-4.1-nano-2025-04-14",
            "openai/gpt-4o-2024-08-06",
            "openai/gpt-4o-2024-11-20",
            "openai/gpt-4o-mini-2024-07-18",
            "openai/gpt-5-2025-08-07",
            "openai/gpt-5-mini-2025-08-07",
            "openai/gpt-5.1-2025-11-13",
            "openai/gpt-5.2-2025-12-11",
            "openai/gpt-5.4-2026-03-05",
            "openai/gpt-5.4-mini-2026-03-17",
            "openai/gpt-5.4-nano-2026-03-17",
            "openai/gpt-5.5",
            "openai/gpt-5.6-luna",
            "openai/gpt-5.6-sol",
            "openai/gpt-5.6-terra",
            "openai/o3-2025-04-16",
            "openai/o3-mini-2025-01-31",
            "openai/o4-mini-2025-04-16",
            "poolside/laguna-m.1",
            "poolside/laguna-xs.2",
            "thinkingmachines/inkling",
            "thinkingmachines/inkling-small",
            "together/meta-llama/Llama-4-Scout-17B-16E-Instruct",
            "together/meta-llama/Meta-Llama-3.1-70B-Instruct-Turbo",
            "together/meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo",
            "together/moonshotai/Kimi-K2-Instruct",
            "xiaomi/mimo-v2.5",
            "xiaomi/mimo-v2.5-pro",
            "zai/glm-4.5",
            "zai/glm-4.6",
            "zai/glm-4.7",
            "zai/glm-5-thinking",
            "zai/glm-5.1",
            "zai/glm-5.2"
        ],
        "partners": [],
        "showBadge": false,
        "visible": true,
        "use_cost_per_test": false,
        "runner": "custom",
        "mode": "one-shot",
        "archived": false,
        "partner": false,
        "total_models": 134
    },
    "tasks": {
        "overall": {
            "anthropic/claude-opus-5": {
                "accuracy": 73.194,
                "latency": 28.573,
                "stderr": 0.871,
                "cost_per_test": 0.860452,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-fable-5": {
                "accuracy": 71.834,
                "latency": 52.99,
                "stderr": 0.883,
                "cost_per_test": 1.737573,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "kimi/kimi-k3": {
                "accuracy": 71.562,
                "latency": 37.99,
                "stderr": 0.888,
                "cost_per_test": 0.063819,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "meta/muse_spark_1_1": {
                "accuracy": 71.29,
                "latency": 27.823,
                "stderr": 0.889,
                "cost_per_test": 0.10923,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "meta/muse_spark_1_2": {
                "accuracy": 70.94,
                "latency": 30.047,
                "stderr": 0.892,
                "cost_per_test": 0.108223,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "thinkingmachines/inkling-small": {
                "accuracy": 69.619,
                "latency": 45.316,
                "stderr": 0.906,
                "cost_per_test": 0.025455,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 262000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "thinkingmachines/inkling": {
                "accuracy": 68.57,
                "latency": 29.199,
                "stderr": 0.915,
                "cost_per_test": 0.084258,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "grok/grok-4.3": {
                "accuracy": 68.532,
                "latency": 36.444,
                "stderr": 0.915,
                "cost_per_test": 0.134161,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 68.415,
                "latency": 23.304,
                "stderr": 0.916,
                "cost_per_test": 0.43166,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 68.259,
                "latency": 80.779,
                "stderr": 0.917,
                "cost_per_test": 0.052412,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "minimax/MiniMax-M3": {
                "accuracy": 68.104,
                "latency": 24.459,
                "stderr": 0.918,
                "cost_per_test": 0.049736,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 512000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "alibaba/qwen3-max-2026-01-23": {
                "accuracy": 68.026,
                "latency": 121.317,
                "stderr": 0.918,
                "cost_per_test": 0.261088,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "grok/grok-4.5": {
                "accuracy": 67.405,
                "latency": 91.881,
                "stderr": 0.924,
                "cost_per_test": 0.221291,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 67.016,
                "latency": 20.581,
                "stderr": 0.926,
                "cost_per_test": 0.487461,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-sonnet-5": {
                "accuracy": 66.977,
                "latency": 18.628,
                "stderr": 0.926,
                "cost_per_test": 0.516119,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-4-fast-reasoning": {
                "accuracy": 66.9,
                "latency": 11.839,
                "stderr": 0.925,
                "cost_per_test": 0.040289,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 66.744,
                "latency": 79.59,
                "stderr": 0.928,
                "cost_per_test": 0.07328,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 48000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "anthropic/claude-opus-4-8": {
                "accuracy": 66.706,
                "latency": 20.226,
                "stderr": 0.928,
                "cost_per_test": 0.855534,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3.6-max-preview": {
                "accuracy": 66.473,
                "latency": 47.068,
                "stderr": 0.929,
                "cost_per_test": 0.156029,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 66.434,
                "latency": 11.279,
                "stderr": 0.93,
                "cost_per_test": 0.047154,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-4.6": {
                "accuracy": 66.161,
                "latency": 24.256,
                "stderr": 0.932,
                "cost_per_test": 0.165532,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "zai/glm-5.2": {
                "accuracy": 66.123,
                "latency": 24.586,
                "stderr": 0.932,
                "cost_per_test": 0.121857,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "anthropic/claude-opus-4-7": {
                "accuracy": 66.084,
                "latency": 17.75,
                "stderr": 0.933,
                "cost_per_test": 0.816017,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-4-0709": {
                "accuracy": 66.045,
                "latency": 29.733,
                "stderr": 0.932,
                "cost_per_test": 0.190665,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 65.967,
                "latency": 28.411,
                "stderr": 0.934,
                "cost_per_test": 0.015643,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 65.889,
                "latency": 26.038,
                "stderr": 0.934,
                "cost_per_test": 0.153756,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3.8-max": {
                "accuracy": 65.851,
                "latency": 46.649,
                "stderr": 0.935,
                "cost_per_test": 0.179323,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "nvidia/nemotron-3-ultra-550b-a55b": {
                "accuracy": 65.462,
                "latency": 15.015,
                "stderr": 0.937,
                "cost_per_test": null,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Nvidia",
                "harness": null
            },
            "deepseek/deepseek-v4-pro-0813": {
                "accuracy": 65.423,
                "latency": 32.854,
                "stderr": 0.937,
                "cost_per_test": 0.002183,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "openai/gpt-5.6-terra": {
                "accuracy": 65.307,
                "latency": 7.717,
                "stderr": 0.938,
                "cost_per_test": 0.171459,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 65.307,
                "latency": 18.483,
                "stderr": 0.938,
                "cost_per_test": 0.307405,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3.5-plus-thinking": {
                "accuracy": 65.307,
                "latency": 155.861,
                "stderr": 0.936,
                "cost_per_test": 0.058073,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 65.268,
                "latency": 82.34,
                "stderr": 0.938,
                "cost_per_test": 0.235266,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "meta/muse_spark": {
                "accuracy": 65.113,
                "latency": 40.057,
                "stderr": 0.936,
                "cost_per_test": 0.011749,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 65.074,
                "latency": 17.725,
                "stderr": 0.939,
                "cost_per_test": 0.173423,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.5-flash": {
                "accuracy": 64.686,
                "latency": 8.7,
                "stderr": 0.942,
                "cost_per_test": 0.140208,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 64.491,
                "latency": 25.388,
                "stderr": 0.942,
                "cost_per_test": 0.181957,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "zai/glm-5.1": {
                "accuracy": 64.452,
                "latency": 68.17,
                "stderr": 0.943,
                "cost_per_test": 0.078922,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "openai/gpt-5.6-sol": {
                "accuracy": 64.375,
                "latency": 15.941,
                "stderr": 0.943,
                "cost_per_test": 0.484982,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.6-luna": {
                "accuracy": 64.219,
                "latency": 16.968,
                "stderr": 0.945,
                "cost_per_test": 0.019039,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 63.831,
                "latency": 35.289,
                "stderr": 0.947,
                "cost_per_test": 0.091991,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3.7-max": {
                "accuracy": 63.714,
                "latency": 19.934,
                "stderr": 0.946,
                "cost_per_test": 0.219883,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 63.675,
                "latency": 9.67,
                "stderr": 0.947,
                "cost_per_test": 0.163386,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 63.675,
                "latency": 20.059,
                "stderr": 0.946,
                "cost_per_test": 0.179069,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3.5-flash": {
                "accuracy": 63.559,
                "latency": 35.198,
                "stderr": 0.947,
                "cost_per_test": 0.009638,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "google/gemini-3.6-flash": {
                "accuracy": 63.326,
                "latency": 6.329,
                "stderr": 0.949,
                "cost_per_test": 0.136842,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-4.1-2025-04-14": {
                "accuracy": 63.054,
                "latency": 10.14,
                "stderr": 0.951,
                "cost_per_test": 0.13202,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-5-thinking": {
                "accuracy": 62.898,
                "latency": 68.577,
                "stderr": 0.95,
                "cost_per_test": 0.067926,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "alibaba/qwen3.6-27b": {
                "accuracy": 62.315,
                "latency": 62.759,
                "stderr": 0.954,
                "cost_per_test": 0.053549,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 61.966,
                "latency": 20.544,
                "stderr": 0.956,
                "cost_per_test": 0.255071,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "ant/ling-3.0-flash-2607": {
                "accuracy": 61.927,
                "latency": 12.451,
                "stderr": 0.957,
                "cost_per_test": 0.006376,
                "temperature": 0.6,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 61.927,
                "latency": 35.839,
                "stderr": 0.957,
                "cost_per_test": 0.047203,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "deepseek/deepseek-v4-flash-0731": {
                "accuracy": 61.849,
                "latency": 16.477,
                "stderr": 0.957,
                "cost_per_test": 0.011801,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "xiaomi/mimo-v2.5-pro": {
                "accuracy": 61.422,
                "latency": 18.899,
                "stderr": 0.959,
                "cost_per_test": 0.037238,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 61.383,
                "latency": 79.331,
                "stderr": 0.959,
                "cost_per_test": 0.148855,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101": {
                "accuracy": 61.305,
                "latency": 9.401,
                "stderr": 0.96,
                "cost_per_test": 0.150159,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514-thinking": {
                "accuracy": 61.228,
                "latency": 168.523,
                "stderr": 0.959,
                "cost_per_test": 0.236746,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 61.189,
                "latency": 9.397,
                "stderr": 0.961,
                "cost_per_test": 0.01612,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 61.189,
                "latency": 10.973,
                "stderr": 0.96,
                "cost_per_test": 0.022663,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "grok/grok-3-mini-fast-high-reasoning": {
                "accuracy": 61.111,
                "latency": 14.386,
                "stderr": 0.957,
                "cost_per_test": 0.017292,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 61.072,
                "latency": 160.705,
                "stderr": 0.961,
                "cost_per_test": 0.086116,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "mistralai/mistral-large-2512": {
                "accuracy": 61.033,
                "latency": 48.842,
                "stderr": 0.96,
                "cost_per_test": 0.024166,
                "temperature": 0.065,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "zai/glm-4.5": {
                "accuracy": 60.956,
                "latency": 95.244,
                "stderr": 0.959,
                "cost_per_test": 0.03783,
                "temperature": 0.6,
                "top_p": 1.0,
                "max_output_tokens": 81920,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 60.917,
                "latency": 52.297,
                "stderr": 0.962,
                "cost_per_test": 0.076368,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-5-20250929": {
                "accuracy": 60.8,
                "latency": 13.383,
                "stderr": 0.962,
                "cost_per_test": 0.237245,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-2.5-pro": {
                "accuracy": 60.8,
                "latency": 27.425,
                "stderr": 0.961,
                "cost_per_test": 0.09535,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.5-flash-lite": {
                "accuracy": 60.684,
                "latency": 3.836,
                "stderr": 0.962,
                "cost_per_test": 0.028548,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 60.606,
                "latency": 14.242,
                "stderr": 0.962,
                "cost_per_test": 0.089265,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "kimi/kimi-k2-thinking": {
                "accuracy": 60.567,
                "latency": 93.297,
                "stderr": 0.962,
                "cost_per_test": 0.038163,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "anthropic/claude-3-7-sonnet-20250219-thinking": {
                "accuracy": 60.412,
                "latency": 22.163,
                "stderr": 0.964,
                "cost_per_test": 0.232197,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 18384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001": {
                "accuracy": 60.295,
                "latency": 7.951,
                "stderr": 0.964,
                "cost_per_test": 0.075779,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 60.179,
                "latency": 73.788,
                "stderr": 0.965,
                "cost_per_test": 0.020976,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "xiaomi/mimo-v2.5": {
                "accuracy": 59.907,
                "latency": 17.625,
                "stderr": 0.966,
                "cost_per_test": 0.011733,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "google/gemini-2.5-pro-exp-03-25": {
                "accuracy": 59.829,
                "latency": 16.302,
                "stderr": 0.966,
                "cost_per_test": 0.08579,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025-thinking": {
                "accuracy": 59.751,
                "latency": 51.757,
                "stderr": 0.966,
                "cost_per_test": 0.026119,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/o3-2025-04-16": {
                "accuracy": 59.713,
                "latency": 19.153,
                "stderr": 0.965,
                "cost_per_test": 0.133487,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-3": {
                "accuracy": 59.713,
                "latency": 24.354,
                "stderr": 0.965,
                "cost_per_test": 0.169809,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "minimax/MiniMax-M2.5": {
                "accuracy": 59.596,
                "latency": 16.599,
                "stderr": 0.966,
                "cost_per_test": 0.024284,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "grok/grok-3-mini-fast-low-reasoning": {
                "accuracy": 59.479,
                "latency": 10.516,
                "stderr": 0.96,
                "cost_per_test": 0.01706,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "low",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 59.363,
                "latency": 4.03,
                "stderr": 0.967,
                "cost_per_test": 0.022927,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/o4-mini-2025-04-16": {
                "accuracy": 58.974,
                "latency": 17.528,
                "stderr": 0.967,
                "cost_per_test": 0.074104,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025": {
                "accuracy": 58.974,
                "latency": 51.895,
                "stderr": 0.969,
                "cost_per_test": 0.026123,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "minimax/MiniMax-M2.1": {
                "accuracy": 58.896,
                "latency": 21.969,
                "stderr": 0.968,
                "cost_per_test": 0.022461,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "mistralai/mistral-medium-3.5": {
                "accuracy": 58.78,
                "latency": 43.699,
                "stderr": 0.969,
                "cost_per_test": 0.14574,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "grok/grok-4-fast-non-reasoning": {
                "accuracy": 58.392,
                "latency": 9.37,
                "stderr": 0.967,
                "cost_per_test": 0.039974,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "fireworks/gpt-oss-120b": {
                "accuracy": 58.236,
                "latency": 42.722,
                "stderr": 0.968,
                "cost_per_test": 0.009601,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "poolside/laguna-m.1": {
                "accuracy": 58.159,
                "latency": 211.713,
                "stderr": 0.971,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "openai/gpt-4.1-mini-2025-04-14": {
                "accuracy": 57.926,
                "latency": 6.443,
                "stderr": 0.972,
                "cost_per_test": 0.026113,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025-thinking": {
                "accuracy": 57.576,
                "latency": 54.225,
                "stderr": 0.973,
                "cost_per_test": 0.008596,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "zai/glm-4.6": {
                "accuracy": 56.838,
                "latency": 78.061,
                "stderr": 0.97,
                "cost_per_test": 0.04461,
                "temperature": 0.6,
                "top_p": 1.0,
                "max_output_tokens": 122880,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "poolside/laguna-xs.2": {
                "accuracy": 56.333,
                "latency": 188.244,
                "stderr": 0.976,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025": {
                "accuracy": 56.294,
                "latency": 54.75,
                "stderr": 0.976,
                "cost_per_test": 0.008646,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3-max": {
                "accuracy": 55.944,
                "latency": 55.675,
                "stderr": 0.977,
                "cost_per_test": 0.086893,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "fireworks/deepseek-v3-0324": {
                "accuracy": 54.74,
                "latency": 36.732,
                "stderr": 0.978,
                "cost_per_test": 0.052049,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514": {
                "accuracy": 54.701,
                "latency": 186.345,
                "stderr": 0.98,
                "cost_per_test": 0.227257,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "arcee-ai/trinity-large-thinking": {
                "accuracy": 54.662,
                "latency": 12.482,
                "stderr": 0.977,
                "cost_per_test": 0.01939,
                "temperature": 0.3,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Arcee-Ai",
                "harness": null
            },
            "fireworks/nemotron-lightning-3p5-30b-a3b": {
                "accuracy": 54.235,
                "latency": 38.792,
                "stderr": 0.98,
                "cost_per_test": 0.004794,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-04-17": {
                "accuracy": 54.157,
                "latency": 8.992,
                "stderr": 0.982,
                "cost_per_test": 0.020528,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "fireworks/deepseek-r1": {
                "accuracy": 54.118,
                "latency": 46.774,
                "stderr": 0.979,
                "cost_per_test": 0.086732,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "anthropic/claude-3-5-sonnet-20241022": {
                "accuracy": 53.613,
                "latency": 0.049,
                "stderr": 0.981,
                "cost_per_test": 0.23772,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "fireworks/gpt-oss-20b": {
                "accuracy": 53.147,
                "latency": 33.968,
                "stderr": 0.979,
                "cost_per_test": 0.004549,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "alibaba/qwen3-max-preview": {
                "accuracy": 52.953,
                "latency": 227.451,
                "stderr": 0.975,
                "cost_per_test": 0.09654,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "grok/grok-4-1-fast-non-reasoning": {
                "accuracy": 52.486,
                "latency": 13.465,
                "stderr": 0.971,
                "cost_per_test": 0.015705,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "fireworks/deepseek-v3": {
                "accuracy": 52.486,
                "latency": 28.322,
                "stderr": 0.98,
                "cost_per_test": 0.052026,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "fireworks/deepseek-v3p1": {
                "accuracy": 51.476,
                "latency": 62.999,
                "stderr": 0.982,
                "cost_per_test": 0.037369,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "grok/grok-2-1212": {
                "accuracy": 51.126,
                "latency": 87.134,
                "stderr": 0.981,
                "cost_per_test": 0.115097,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 0,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "fireworks/deepseek-v3p2-thinking": {
                "accuracy": 50.971,
                "latency": 99.485,
                "stderr": 0.979,
                "cost_per_test": 0.038215,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 20480,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "anthropic/claude-3-5-haiku-20241022": {
                "accuracy": 50.816,
                "latency": 0.039,
                "stderr": 0.981,
                "cost_per_test": 0.063315,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "mistralai/mistral-medium-2505": {
                "accuracy": 50.738,
                "latency": 30.346,
                "stderr": 0.974,
                "cost_per_test": 0.024082,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "together/moonshotai/Kimi-K2-Instruct": {
                "accuracy": 50.388,
                "latency": 40.894,
                "stderr": 0.985,
                "cost_per_test": 0.059336,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "fireworks/llama4-maverick-instruct-basic": {
                "accuracy": 49.728,
                "latency": 4.724,
                "stderr": 0.981,
                "cost_per_test": 0.014178,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "fireworks/deepseek-v3p2": {
                "accuracy": 47.941,
                "latency": 133.787,
                "stderr": 0.98,
                "cost_per_test": 0.036826,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 20480,
                "reasoning": null,
                "reasoning_effort": "none",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "mistralai/magistral-medium-2509": {
                "accuracy": 47.397,
                "latency": 55.265,
                "stderr": 0.982,
                "cost_per_test": 0.125244,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "together/meta-llama/Llama-4-Scout-17B-16E-Instruct": {
                "accuracy": 46.776,
                "latency": 9.697,
                "stderr": 0.98,
                "cost_per_test": 0.011639,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "zai/glm-4.7": {
                "accuracy": 46.387,
                "latency": 66.101,
                "stderr": 0.899,
                "cost_per_test": 0.039082,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "cohere/command-a-03-2025": {
                "accuracy": 45.96,
                "latency": 14.146,
                "stderr": 0.972,
                "cost_per_test": 0.206138,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 8000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "openai/gpt-4o-2024-11-20": {
                "accuracy": 45.921,
                "latency": 6.461,
                "stderr": 0.973,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-4o-mini-2024-07-18": {
                "accuracy": 45.455,
                "latency": 0.041,
                "stderr": 0.977,
                "cost_per_test": 0.008848,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/o3-mini-2025-01-31": {
                "accuracy": 45.299,
                "latency": 31.047,
                "stderr": 0.964,
                "cost_per_test": 0.078121,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "mistralai/mistral-small-2503": {
                "accuracy": 44.173,
                "latency": 11.291,
                "stderr": 0.972,
                "cost_per_test": 0.004408,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "mistralai/magistral-small-2509": {
                "accuracy": 44.017,
                "latency": 14.053,
                "stderr": 0.965,
                "cost_per_test": 0.030437,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "google/gemini-2.0-pro-exp-02-05": {
                "accuracy": 43.435,
                "latency": 18.879,
                "stderr": 0.971,
                "cost_per_test": 0.104585,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-4.1-nano-2025-04-14": {
                "accuracy": 42.075,
                "latency": 4.954,
                "stderr": 0.968,
                "cost_per_test": 0.006454,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "ai21labs/jamba-large-1.6": {
                "accuracy": 41.531,
                "latency": 27.38,
                "stderr": 0.963,
                "cost_per_test": 0.150029,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "google/gemini-1.5-pro-002": {
                "accuracy": 40.521,
                "latency": 38.05,
                "stderr": 0.967,
                "cost_per_test": 0.131149,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-4o-2024-08-06": {
                "accuracy": 39.433,
                "latency": 0.046,
                "stderr": 0.951,
                "cost_per_test": 0.147391,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "ai21labs/jamba-1.5-large": {
                "accuracy": 39.433,
                "latency": 10.765,
                "stderr": 0.959,
                "cost_per_test": 0.149067,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "together/meta-llama/Meta-Llama-3.1-70B-Instruct-Turbo": {
                "accuracy": 38.85,
                "latency": 0.062,
                "stderr": 0.95,
                "cost_per_test": 0.053099,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "google/gemini-1.5-flash-002": {
                "accuracy": 38.19,
                "latency": 28.407,
                "stderr": 0.958,
                "cost_per_test": 0.007862,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "ai21labs/jamba-mini-1.6": {
                "accuracy": 38.034,
                "latency": 4.248,
                "stderr": 0.951,
                "cost_per_test": 0.014897,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "together/meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo": {
                "accuracy": 37.801,
                "latency": 0.038,
                "stderr": 0.934,
                "cost_per_test": 0.010777,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "ai21labs/jamba-1.5-mini": {
                "accuracy": 33.877,
                "latency": 2.315,
                "stderr": 0.929,
                "cost_per_test": 0.014865,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "google/gemini-2.0-flash-001": {
                "accuracy": 33.722,
                "latency": 30.038,
                "stderr": 0.93,
                "cost_per_test": 0.010481,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-1.5-flash-001": {
                "accuracy": 28.633,
                "latency": 1.156,
                "stderr": 0.88,
                "cost_per_test": 0.006413,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            }
        },
        "exact_pages": {
            "kimi/kimi-k3": {
                "accuracy": 69.347,
                "latency": 19.173,
                "stderr": 1.574,
                "cost_per_test": 0.012378,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "thinkingmachines/inkling": {
                "accuracy": 68.881,
                "latency": 18.519,
                "stderr": 1.581,
                "cost_per_test": 0.006684,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "anthropic/claude-opus-5": {
                "accuracy": 68.648,
                "latency": 16.028,
                "stderr": 1.584,
                "cost_per_test": 0.044928,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "thinkingmachines/inkling-small": {
                "accuracy": 68.648,
                "latency": 38.507,
                "stderr": 1.584,
                "cost_per_test": 0.001897,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 262000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 68.065,
                "latency": 11.684,
                "stderr": 1.592,
                "cost_per_test": 0.098516,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 68.065,
                "latency": 33.05,
                "stderr": 1.592,
                "cost_per_test": 0.335055,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4-fast-reasoning": {
                "accuracy": 67.249,
                "latency": 8.663,
                "stderr": 1.602,
                "cost_per_test": 0.001343,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "grok/grok-3-mini-fast-low-reasoning": {
                "accuracy": 67.016,
                "latency": 5.678,
                "stderr": 1.605,
                "cost_per_test": 0.000896,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "low",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-fable-5": {
                "accuracy": 66.9,
                "latency": 18.601,
                "stderr": 1.607,
                "cost_per_test": 0.090437,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-4.3": {
                "accuracy": 66.783,
                "latency": 45.867,
                "stderr": 1.608,
                "cost_per_test": 0.004871,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "meta/muse_spark_1_1": {
                "accuracy": 66.434,
                "latency": 18.683,
                "stderr": 1.612,
                "cost_per_test": 0.009465,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "grok/grok-4.5": {
                "accuracy": 66.317,
                "latency": 154.026,
                "stderr": 1.614,
                "cost_per_test": 0.007543,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 66.2,
                "latency": 40.528,
                "stderr": 1.615,
                "cost_per_test": 0.004139,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "meta/muse_spark_1_2": {
                "accuracy": 66.084,
                "latency": 17.558,
                "stderr": 1.616,
                "cost_per_test": 0.006998,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "grok/grok-4-0709": {
                "accuracy": 65.967,
                "latency": 11.944,
                "stderr": 1.618,
                "cost_per_test": 0.021639,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "grok/grok-3-mini-fast-high-reasoning": {
                "accuracy": 65.851,
                "latency": 7.748,
                "stderr": 1.619,
                "cost_per_test": 0.001098,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 65.618,
                "latency": 17.821,
                "stderr": 1.622,
                "cost_per_test": 0.02963,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 65.501,
                "latency": 32.826,
                "stderr": 1.623,
                "cost_per_test": 0.033844,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 64.918,
                "latency": 10.405,
                "stderr": 1.629,
                "cost_per_test": 0.022285,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "minimax/MiniMax-M3": {
                "accuracy": 64.918,
                "latency": 18.56,
                "stderr": 1.629,
                "cost_per_test": 0.002351,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 512000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "alibaba/qwen3-max-2026-01-23": {
                "accuracy": 64.918,
                "latency": 76.495,
                "stderr": 1.629,
                "cost_per_test": 0.018879,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "grok/grok-4-fast-non-reasoning": {
                "accuracy": 64.802,
                "latency": 6.835,
                "stderr": 1.63,
                "cost_per_test": 0.001137,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 64.802,
                "latency": 38.306,
                "stderr": 1.63,
                "cost_per_test": 0.006637,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 48000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "meta/muse_spark": {
                "accuracy": 64.802,
                "latency": 41.977,
                "stderr": 1.63,
                "cost_per_test": 0.001586,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "alibaba/qwen3.8-max": {
                "accuracy": 64.569,
                "latency": 23.674,
                "stderr": 1.633,
                "cost_per_test": 0.009952,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "grok/grok-4.6": {
                "accuracy": 64.452,
                "latency": 15.54,
                "stderr": 1.634,
                "cost_per_test": 0.008411,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "nvidia/nemotron-3-ultra-550b-a55b": {
                "accuracy": 64.336,
                "latency": 6.738,
                "stderr": 1.635,
                "cost_per_test": null,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Nvidia",
                "harness": null
            },
            "anthropic/claude-opus-4-7": {
                "accuracy": 64.336,
                "latency": 10.524,
                "stderr": 1.635,
                "cost_per_test": 0.03284,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-4-8": {
                "accuracy": 64.336,
                "latency": 14.109,
                "stderr": 1.635,
                "cost_per_test": 0.039715,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "zai/glm-4.5": {
                "accuracy": 64.336,
                "latency": 57.11,
                "stderr": 1.635,
                "cost_per_test": 0.002725,
                "temperature": 0.6,
                "top_p": 1.0,
                "max_output_tokens": 81920,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 64.103,
                "latency": 51.952,
                "stderr": 1.638,
                "cost_per_test": 0.025446,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/o4-mini-2025-04-16": {
                "accuracy": 63.986,
                "latency": 8.361,
                "stderr": 1.639,
                "cost_per_test": 0.00449,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-sonnet-5": {
                "accuracy": 63.869,
                "latency": 12.323,
                "stderr": 1.64,
                "cost_per_test": 0.027074,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "zai/glm-5.2": {
                "accuracy": 63.636,
                "latency": 12.744,
                "stderr": 1.642,
                "cost_per_test": 0.005283,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "ant/ling-3.0-flash-2607": {
                "accuracy": 63.52,
                "latency": 3.526,
                "stderr": 1.643,
                "cost_per_test": 0.000229,
                "temperature": 0.6,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "openai/o3-2025-04-16": {
                "accuracy": 63.52,
                "latency": 10.487,
                "stderr": 1.643,
                "cost_per_test": 0.008451,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.6-luna": {
                "accuracy": 63.52,
                "latency": 10.874,
                "stderr": 1.643,
                "cost_per_test": 0.001987,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 63.403,
                "latency": 9.631,
                "stderr": 1.644,
                "cost_per_test": 0.020493,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "deepseek/deepseek-v4-pro-0813": {
                "accuracy": 63.287,
                "latency": 26.006,
                "stderr": 1.646,
                "cost_per_test": 0.001529,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "alibaba/qwen3.6-max-preview": {
                "accuracy": 63.287,
                "latency": 42.169,
                "stderr": 1.646,
                "cost_per_test": 0.013772,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/gpt-5.6-terra": {
                "accuracy": 63.17,
                "latency": 4.35,
                "stderr": 1.647,
                "cost_per_test": 0.007123,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 63.17,
                "latency": 9.373,
                "stderr": 1.647,
                "cost_per_test": 0.014257,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 63.054,
                "latency": 3.81,
                "stderr": 1.648,
                "cost_per_test": 0.008149,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 63.054,
                "latency": 24.022,
                "stderr": 1.648,
                "cost_per_test": 0.004427,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-2.5-pro": {
                "accuracy": 62.937,
                "latency": 32.815,
                "stderr": 1.649,
                "cost_per_test": 0.201351,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-4.1-2025-04-14": {
                "accuracy": 62.821,
                "latency": 3.975,
                "stderr": 1.65,
                "cost_per_test": 0.005077,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3.5-flash": {
                "accuracy": 62.821,
                "latency": 5.149,
                "stderr": 1.65,
                "cost_per_test": 0.011121,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "fireworks/gpt-oss-120b": {
                "accuracy": 61.888,
                "latency": 46.794,
                "stderr": 1.658,
                "cost_per_test": 0.000868,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "zai/glm-5.1": {
                "accuracy": 61.772,
                "latency": 18.86,
                "stderr": 1.659,
                "cost_per_test": 0.005086,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "minimax/MiniMax-M2.5": {
                "accuracy": 61.655,
                "latency": 10.304,
                "stderr": 1.66,
                "cost_per_test": 0.002217,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "openai/gpt-5.6-sol": {
                "accuracy": 61.538,
                "latency": 8.564,
                "stderr": 1.661,
                "cost_per_test": 0.026875,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "mistralai/mistral-medium-2505": {
                "accuracy": 61.538,
                "latency": 11.934,
                "stderr": 1.661,
                "cost_per_test": 0.00108,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 61.538,
                "latency": 44.668,
                "stderr": 1.661,
                "cost_per_test": 0.014516,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "minimax/MiniMax-M2.1": {
                "accuracy": 61.305,
                "latency": 15.39,
                "stderr": 1.663,
                "cost_per_test": 0.001629,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "google/gemini-3.6-flash": {
                "accuracy": 61.072,
                "latency": 4.299,
                "stderr": 1.665,
                "cost_per_test": 0.008536,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "poolside/laguna-m.1": {
                "accuracy": 60.956,
                "latency": 190.749,
                "stderr": 1.665,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "deepseek/deepseek-v4-flash-0731": {
                "accuracy": 60.839,
                "latency": 9.468,
                "stderr": 1.666,
                "cost_per_test": 0.000488,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101": {
                "accuracy": 60.839,
                "latency": 11.462,
                "stderr": 1.666,
                "cost_per_test": 0.007209,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 60.839,
                "latency": 19.092,
                "stderr": 1.666,
                "cost_per_test": 0.012811,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3.5-plus-thinking": {
                "accuracy": 60.839,
                "latency": 200.413,
                "stderr": 1.666,
                "cost_per_test": 0.006714,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "alibaba/qwen3.5-flash": {
                "accuracy": 60.723,
                "latency": 13.893,
                "stderr": 1.667,
                "cost_per_test": 0.00112,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 60.606,
                "latency": 12.18,
                "stderr": 1.668,
                "cost_per_test": 0.01402,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 60.606,
                "latency": 28.533,
                "stderr": 1.668,
                "cost_per_test": 0.00541,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 60.373,
                "latency": 8.67,
                "stderr": 1.67,
                "cost_per_test": 0.000735,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-5-20250929": {
                "accuracy": 60.373,
                "latency": 9.701,
                "stderr": 1.67,
                "cost_per_test": 0.007791,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "xiaomi/mimo-v2.5-pro": {
                "accuracy": 60.373,
                "latency": 11.967,
                "stderr": 1.67,
                "cost_per_test": 0.00107,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "fireworks/deepseek-v3-0324": {
                "accuracy": 60.373,
                "latency": 12.461,
                "stderr": 1.67,
                "cost_per_test": 0.001418,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001": {
                "accuracy": 60.256,
                "latency": 5.204,
                "stderr": 1.671,
                "cost_per_test": 0.002564,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 60.256,
                "latency": 7.227,
                "stderr": 1.671,
                "cost_per_test": 0.00105,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "zai/glm-5-thinking": {
                "accuracy": 60.14,
                "latency": 31.137,
                "stderr": 1.672,
                "cost_per_test": 0.006603,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "arcee-ai/trinity-large-thinking": {
                "accuracy": 60.023,
                "latency": 9.333,
                "stderr": 1.672,
                "cost_per_test": 0.000879,
                "temperature": 0.3,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Arcee-Ai",
                "harness": null
            },
            "alibaba/qwen3.7-max": {
                "accuracy": 60.023,
                "latency": 15.358,
                "stderr": 1.672,
                "cost_per_test": 0.01138,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "zai/glm-4.7": {
                "accuracy": 60.023,
                "latency": 39.216,
                "stderr": 1.672,
                "cost_per_test": 0.00438,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "alibaba/qwen3-max-preview": {
                "accuracy": 59.907,
                "latency": 5.561,
                "stderr": 1.673,
                "cost_per_test": 0.003066,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "google/gemini-2.5-pro-exp-03-25": {
                "accuracy": 59.907,
                "latency": 16.774,
                "stderr": 1.673,
                "cost_per_test": 0.003881,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3-max": {
                "accuracy": 59.79,
                "latency": 7.092,
                "stderr": 1.674,
                "cost_per_test": 0.002796,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "anthropic/claude-3-7-sonnet-20250219-thinking": {
                "accuracy": 59.79,
                "latency": 10.947,
                "stderr": 1.674,
                "cost_per_test": 0.01504,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 18384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "mistralai/mistral-large-2512": {
                "accuracy": 59.79,
                "latency": 20.072,
                "stderr": 1.674,
                "cost_per_test": 0.035945,
                "temperature": 0.065,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 59.79,
                "latency": 148.152,
                "stderr": 1.674,
                "cost_per_test": 0.003682,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3.6-27b": {
                "accuracy": 59.674,
                "latency": 27.389,
                "stderr": 1.675,
                "cost_per_test": 0.006658,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "fireworks/deepseek-r1": {
                "accuracy": 59.557,
                "latency": 20.406,
                "stderr": 1.675,
                "cost_per_test": 0.00529,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "poolside/laguna-xs.2": {
                "accuracy": 59.557,
                "latency": 146.965,
                "stderr": 1.675,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "openai/gpt-4.1-mini-2025-04-14": {
                "accuracy": 59.441,
                "latency": 3.307,
                "stderr": 1.676,
                "cost_per_test": 0.000891,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "xiaomi/mimo-v2.5": {
                "accuracy": 59.441,
                "latency": 5.643,
                "stderr": 1.676,
                "cost_per_test": 0.00025,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 59.441,
                "latency": 14.982,
                "stderr": 1.676,
                "cost_per_test": 0.005289,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 59.207,
                "latency": 84.032,
                "stderr": 1.678,
                "cost_per_test": 0.007781,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "kimi/kimi-k2-thinking": {
                "accuracy": 59.091,
                "latency": 67.814,
                "stderr": 1.679,
                "cost_per_test": 0.003367,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "fireworks/gpt-oss-20b": {
                "accuracy": 58.974,
                "latency": 32.257,
                "stderr": 1.679,
                "cost_per_test": 0.000678,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 58.858,
                "latency": 17.494,
                "stderr": 1.68,
                "cost_per_test": 0.003881,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "fireworks/deepseek-v3": {
                "accuracy": 58.741,
                "latency": 6.717,
                "stderr": 1.681,
                "cost_per_test": 0.001386,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514-thinking": {
                "accuracy": 58.741,
                "latency": 35.19,
                "stderr": 1.681,
                "cost_per_test": 0.014194,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 58.625,
                "latency": 9.317,
                "stderr": 1.681,
                "cost_per_test": 0.004717,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514": {
                "accuracy": 58.508,
                "latency": 5.2,
                "stderr": 1.682,
                "cost_per_test": 0.007941,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-3": {
                "accuracy": 58.508,
                "latency": 5.488,
                "stderr": 1.682,
                "cost_per_test": 0.007219,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "fireworks/nemotron-lightning-3p5-30b-a3b": {
                "accuracy": 58.508,
                "latency": 27.393,
                "stderr": 1.682,
                "cost_per_test": 0.000497,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "zai/glm-4.6": {
                "accuracy": 58.159,
                "latency": 15.911,
                "stderr": 1.684,
                "cost_per_test": 0.003792,
                "temperature": 0.6,
                "top_p": 1.0,
                "max_output_tokens": 122880,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "anthropic/claude-3-5-sonnet-20241022": {
                "accuracy": 57.809,
                "latency": 0.047,
                "stderr": 1.686,
                "cost_per_test": 0.007112,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.5-flash-lite": {
                "accuracy": 57.459,
                "latency": 2.557,
                "stderr": 1.688,
                "cost_per_test": 0.002608,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-2-1212": {
                "accuracy": 57.226,
                "latency": 3.662,
                "stderr": 1.689,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 0,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "fireworks/deepseek-v3p1": {
                "accuracy": 56.876,
                "latency": 31.727,
                "stderr": 1.691,
                "cost_per_test": 0.000982,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025-thinking": {
                "accuracy": 56.76,
                "latency": 57.491,
                "stderr": 1.691,
                "cost_per_test": 0.00088,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 56.643,
                "latency": 3.379,
                "stderr": 1.692,
                "cost_per_test": 0.001726,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-3-5-haiku-20241022": {
                "accuracy": 56.41,
                "latency": 0.062,
                "stderr": 1.693,
                "cost_per_test": 0.001851,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/o3-mini-2025-01-31": {
                "accuracy": 56.41,
                "latency": 12.114,
                "stderr": 1.693,
                "cost_per_test": 0.004987,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "cohere/command-a-03-2025": {
                "accuracy": 56.177,
                "latency": 3.828,
                "stderr": 1.694,
                "cost_per_test": 0.005637,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 8000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025": {
                "accuracy": 56.177,
                "latency": 57.467,
                "stderr": 1.694,
                "cost_per_test": 0.000911,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "fireworks/llama4-maverick-instruct-basic": {
                "accuracy": 55.828,
                "latency": 3.167,
                "stderr": 1.695,
                "cost_per_test": 0.000535,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "mistralai/magistral-small-2509": {
                "accuracy": 55.478,
                "latency": 8.521,
                "stderr": 1.697,
                "cost_per_test": 0.002287,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "mistralai/mistral-medium-3.5": {
                "accuracy": 55.361,
                "latency": 30.593,
                "stderr": 1.697,
                "cost_per_test": 0.018845,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "openai/gpt-4o-2024-11-20": {
                "accuracy": 55.245,
                "latency": 2.215,
                "stderr": 1.698,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-04-17": {
                "accuracy": 54.779,
                "latency": 1.657,
                "stderr": 1.699,
                "cost_per_test": 0.000864,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025-thinking": {
                "accuracy": 53.846,
                "latency": 57.443,
                "stderr": 1.702,
                "cost_per_test": 0.000212,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025": {
                "accuracy": 53.73,
                "latency": 59.381,
                "stderr": 1.702,
                "cost_per_test": 0.000222,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "together/meta-llama/Llama-4-Scout-17B-16E-Instruct": {
                "accuracy": 52.681,
                "latency": 2.476,
                "stderr": 1.705,
                "cost_per_test": 0.000367,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "mistralai/mistral-small-2503": {
                "accuracy": 52.448,
                "latency": 8.023,
                "stderr": 1.705,
                "cost_per_test": 0.000173,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "openai/gpt-4o-mini-2024-07-18": {
                "accuracy": 52.214,
                "latency": 0.012,
                "stderr": 1.705,
                "cost_per_test": 0.000268,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "mistralai/magistral-medium-2509": {
                "accuracy": 51.981,
                "latency": 16.96,
                "stderr": 1.706,
                "cost_per_test": 0.011202,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "google/gemini-2.0-pro-exp-02-05": {
                "accuracy": 51.166,
                "latency": 5.275,
                "stderr": 1.707,
                "cost_per_test": 0.002429,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "together/meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo": {
                "accuracy": 50.816,
                "latency": 0.012,
                "stderr": 1.707,
                "cost_per_test": 0.000288,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "openai/gpt-4o-2024-08-06": {
                "accuracy": 50.233,
                "latency": 0.073,
                "stderr": 1.707,
                "cost_per_test": 0.004761,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "ai21labs/jamba-large-1.6": {
                "accuracy": 50.117,
                "latency": 7.516,
                "stderr": 1.707,
                "cost_per_test": 0.0044,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "together/moonshotai/Kimi-K2-Instruct": {
                "accuracy": 49.417,
                "latency": 46.214,
                "stderr": 1.707,
                "cost_per_test": 0.111403,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "openai/gpt-4.1-nano-2025-04-14": {
                "accuracy": 48.951,
                "latency": 3.869,
                "stderr": 1.707,
                "cost_per_test": 0.000183,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "together/meta-llama/Meta-Llama-3.1-70B-Instruct-Turbo": {
                "accuracy": 48.252,
                "latency": 0.073,
                "stderr": 1.706,
                "cost_per_test": 0.001373,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "grok/grok-4-1-fast-non-reasoning": {
                "accuracy": 46.853,
                "latency": 19.906,
                "stderr": 1.704,
                "cost_per_test": 0.03445,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "ai21labs/jamba-1.5-large": {
                "accuracy": 45.455,
                "latency": 3.138,
                "stderr": 1.7,
                "cost_per_test": 0.003786,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "ai21labs/jamba-mini-1.6": {
                "accuracy": 44.988,
                "latency": 1.553,
                "stderr": 1.698,
                "cost_per_test": 0.000358,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "fireworks/deepseek-v3p2-thinking": {
                "accuracy": 44.755,
                "latency": 173.546,
                "stderr": 1.698,
                "cost_per_test": 0.075006,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 20480,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "fireworks/deepseek-v3p2": {
                "accuracy": 43.473,
                "latency": 355.63,
                "stderr": 1.692,
                "cost_per_test": 0.073712,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 20480,
                "reasoning": null,
                "reasoning_effort": "none",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "google/gemini-1.5-pro-002": {
                "accuracy": 40.676,
                "latency": 20.551,
                "stderr": 1.677,
                "cost_per_test": 0.082015,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "ai21labs/jamba-1.5-mini": {
                "accuracy": 39.627,
                "latency": 0.655,
                "stderr": 1.67,
                "cost_per_test": 0.000336,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "google/gemini-1.5-flash-001": {
                "accuracy": 38.462,
                "latency": 3.353,
                "stderr": 1.661,
                "cost_per_test": 0.00014,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-1.5-flash-002": {
                "accuracy": 37.762,
                "latency": 13.387,
                "stderr": 1.655,
                "cost_per_test": 0.004915,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-2.0-flash-001": {
                "accuracy": 35.431,
                "latency": 18.434,
                "stderr": 1.633,
                "cost_per_test": 0.006552,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            }
        },
        "max_fitting_context": {
            "anthropic/claude-fable-5": {
                "accuracy": 77.04,
                "latency": 81.472,
                "stderr": 1.436,
                "cost_per_test": 3.734685,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-5": {
                "accuracy": 75.991,
                "latency": 40.655,
                "stderr": 1.458,
                "cost_per_test": 1.839482,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "meta/muse_spark_1_1": {
                "accuracy": 74.359,
                "latency": 39.377,
                "stderr": 1.491,
                "cost_per_test": 0.230768,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "meta/muse_spark_1_2": {
                "accuracy": 74.009,
                "latency": 41.511,
                "stderr": 1.497,
                "cost_per_test": 0.230884,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "kimi/kimi-k3": {
                "accuracy": 72.727,
                "latency": 64.691,
                "stderr": 1.52,
                "cost_per_test": 0.122828,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "alibaba/qwen3-max-2026-01-23": {
                "accuracy": 70.629,
                "latency": 198.506,
                "stderr": 1.555,
                "cost_per_test": 0.565658,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "thinkingmachines/inkling-small": {
                "accuracy": 70.396,
                "latency": 45.52,
                "stderr": 1.558,
                "cost_per_test": 0.052717,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 262000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 70.047,
                "latency": 25.327,
                "stderr": 1.564,
                "cost_per_test": 0.902658,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-4-8": {
                "accuracy": 69.814,
                "latency": 26.189,
                "stderr": 1.567,
                "cost_per_test": 1.847531,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "minimax/MiniMax-M3": {
                "accuracy": 69.697,
                "latency": 31.268,
                "stderr": 1.569,
                "cost_per_test": 0.105284,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 512000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "anthropic/claude-sonnet-5": {
                "accuracy": 69.114,
                "latency": 23.105,
                "stderr": 1.577,
                "cost_per_test": 1.105176,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-4.3": {
                "accuracy": 68.765,
                "latency": 38.739,
                "stderr": 1.582,
                "cost_per_test": 0.316319,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "zai/glm-5.2": {
                "accuracy": 68.415,
                "latency": 34.665,
                "stderr": 1.587,
                "cost_per_test": 0.262089,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 68.415,
                "latency": 132.479,
                "stderr": 1.587,
                "cost_per_test": 0.109485,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "alibaba/qwen3.6-max-preview": {
                "accuracy": 68.182,
                "latency": 47.888,
                "stderr": 1.59,
                "cost_per_test": 0.351906,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "anthropic/claude-opus-4-7": {
                "accuracy": 67.832,
                "latency": 24.506,
                "stderr": 1.595,
                "cost_per_test": 1.760496,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "thinkingmachines/inkling": {
                "accuracy": 67.599,
                "latency": 23.995,
                "stderr": 1.598,
                "cost_per_test": 0.173886,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "grok/grok-4.5": {
                "accuracy": 67.483,
                "latency": 109.542,
                "stderr": 1.599,
                "cost_per_test": 0.524857,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 67.249,
                "latency": 103.491,
                "stderr": 1.602,
                "cost_per_test": 0.171139,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 48000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "alibaba/qwen3.5-plus-thinking": {
                "accuracy": 67.249,
                "latency": 226.284,
                "stderr": 1.602,
                "cost_per_test": 0.133024,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "nvidia/nemotron-3-ultra-550b-a55b": {
                "accuracy": 67.133,
                "latency": 22.607,
                "stderr": 1.604,
                "cost_per_test": null,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Nvidia",
                "harness": null
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 67.133,
                "latency": 28.963,
                "stderr": 1.604,
                "cost_per_test": 0.9616,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 67.016,
                "latency": 21.418,
                "stderr": 1.605,
                "cost_per_test": 0.377404,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 66.9,
                "latency": 28.935,
                "stderr": 1.607,
                "cost_per_test": 0.387473,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.5-flash": {
                "accuracy": 66.667,
                "latency": 12.357,
                "stderr": 1.609,
                "cost_per_test": 0.29559,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.6-terra": {
                "accuracy": 66.317,
                "latency": 9.507,
                "stderr": 1.614,
                "cost_per_test": 0.369227,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-5.1": {
                "accuracy": 66.317,
                "latency": 139.915,
                "stderr": 1.614,
                "cost_per_test": 0.15991,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "grok/grok-4.6": {
                "accuracy": 66.2,
                "latency": 33.17,
                "stderr": 1.615,
                "cost_per_test": 0.356004,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "deepseek/deepseek-v4-pro-0813": {
                "accuracy": 66.2,
                "latency": 34.411,
                "stderr": 1.615,
                "cost_per_test": 0.002636,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 66.084,
                "latency": 26.038,
                "stderr": 1.616,
                "cost_per_test": 0.620313,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3.7-max": {
                "accuracy": 65.734,
                "latency": 26.109,
                "stderr": 1.62,
                "cost_per_test": 0.473648,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "alibaba/qwen3.5-flash": {
                "accuracy": 65.734,
                "latency": 57.235,
                "stderr": 1.62,
                "cost_per_test": 0.019801,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 65.618,
                "latency": 23.184,
                "stderr": 1.622,
                "cost_per_test": 0.024271,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3.8-max": {
                "accuracy": 65.618,
                "latency": 60.398,
                "stderr": 1.622,
                "cost_per_test": 0.381993,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "google/gemini-3.6-flash": {
                "accuracy": 65.385,
                "latency": 7.954,
                "stderr": 1.624,
                "cost_per_test": 0.291436,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.6-sol": {
                "accuracy": 65.152,
                "latency": 18.943,
                "stderr": 1.627,
                "cost_per_test": 1.060797,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 64.918,
                "latency": 72.418,
                "stderr": 1.629,
                "cost_per_test": 0.483226,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3.6-27b": {
                "accuracy": 64.918,
                "latency": 109.112,
                "stderr": 1.629,
                "cost_per_test": 0.107726,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 64.802,
                "latency": 33.846,
                "stderr": 1.63,
                "cost_per_test": 0.000351,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "grok/grok-4-1-fast-non-reasoning": {
                "accuracy": 64.103,
                "latency": 12.831,
                "stderr": 1.638,
                "cost_per_test": 0.000347,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-5.6-luna": {
                "accuracy": 64.103,
                "latency": 19.928,
                "stderr": 1.638,
                "cost_per_test": 0.03887,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 63.869,
                "latency": 14.006,
                "stderr": 1.64,
                "cost_per_test": 0.016663,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "mistralai/mistral-large-2512": {
                "accuracy": 63.753,
                "latency": 53.318,
                "stderr": 1.641,
                "cost_per_test": 0.001232,
                "temperature": 0.065,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 63.52,
                "latency": 7.32,
                "stderr": 1.643,
                "cost_per_test": 0.003928,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 63.403,
                "latency": 48.958,
                "stderr": 1.644,
                "cost_per_test": 0.181071,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 63.054,
                "latency": 79.933,
                "stderr": 1.648,
                "cost_per_test": 0.316589,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 62.937,
                "latency": 43.066,
                "stderr": 1.649,
                "cost_per_test": 0.097946,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "grok/grok-4-0709": {
                "accuracy": 62.821,
                "latency": 43.442,
                "stderr": 1.65,
                "cost_per_test": 0.345687,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "deepseek/deepseek-v4-flash-0731": {
                "accuracy": 62.704,
                "latency": 22.531,
                "stderr": 1.651,
                "cost_per_test": 0.02548,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "google/gemini-3.5-flash-lite": {
                "accuracy": 62.354,
                "latency": 4.541,
                "stderr": 1.654,
                "cost_per_test": 0.059711,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-4-fast-reasoning": {
                "accuracy": 62.354,
                "latency": 16.413,
                "stderr": 1.654,
                "cost_per_test": 0.076169,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 62.121,
                "latency": 12.158,
                "stderr": 1.656,
                "cost_per_test": 0.035271,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 62.121,
                "latency": 15.783,
                "stderr": 1.656,
                "cost_per_test": 0.351139,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-5-20250929": {
                "accuracy": 61.538,
                "latency": 16.257,
                "stderr": 1.661,
                "cost_per_test": 0.483633,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "zai/glm-5-thinking": {
                "accuracy": 61.538,
                "latency": 104.077,
                "stderr": 1.661,
                "cost_per_test": 0.123859,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "xiaomi/mimo-v2.5-pro": {
                "accuracy": 61.422,
                "latency": 28.107,
                "stderr": 1.662,
                "cost_per_test": 0.081304,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 61.072,
                "latency": 27.456,
                "stderr": 1.665,
                "cost_per_test": 0.464498,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "mistralai/mistral-medium-3.5": {
                "accuracy": 61.072,
                "latency": 58.363,
                "stderr": 1.665,
                "cost_per_test": 0.291221,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 61.072,
                "latency": 175.462,
                "stderr": 1.665,
                "cost_per_test": 0.175312,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 60.956,
                "latency": 66.943,
                "stderr": 1.665,
                "cost_per_test": 0.146241,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-4.1-2025-04-14": {
                "accuracy": 60.839,
                "latency": 15.579,
                "stderr": 1.666,
                "cost_per_test": 0.261029,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101": {
                "accuracy": 60.723,
                "latency": 5.786,
                "stderr": 1.667,
                "cost_per_test": 0.011333,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 60.606,
                "latency": 4.003,
                "stderr": 1.668,
                "cost_per_test": 0.048652,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025-thinking": {
                "accuracy": 60.606,
                "latency": 67.517,
                "stderr": 1.668,
                "cost_per_test": 0.056863,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025": {
                "accuracy": 60.49,
                "latency": 69.3,
                "stderr": 1.669,
                "cost_per_test": 0.056875,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 60.49,
                "latency": 93.929,
                "stderr": 1.669,
                "cost_per_test": 0.039253,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514-thinking": {
                "accuracy": 60.49,
                "latency": 351.337,
                "stderr": 1.669,
                "cost_per_test": 0.463634,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "xiaomi/mimo-v2.5": {
                "accuracy": 60.14,
                "latency": 27.898,
                "stderr": 1.672,
                "cost_per_test": 0.026197,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 60.023,
                "latency": 18.36,
                "stderr": 1.672,
                "cost_per_test": 0.166751,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "ant/ling-3.0-flash-2607": {
                "accuracy": 60.023,
                "latency": 20.97,
                "stderr": 1.672,
                "cost_per_test": 0.013642,
                "temperature": 0.6,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "meta/muse_spark": {
                "accuracy": 60.023,
                "latency": 38.935,
                "stderr": 1.672,
                "cost_per_test": 0.022947,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025-thinking": {
                "accuracy": 60.023,
                "latency": 70.079,
                "stderr": 1.672,
                "cost_per_test": 0.01883,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025": {
                "accuracy": 59.907,
                "latency": 70.729,
                "stderr": 1.673,
                "cost_per_test": 0.018892,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 59.557,
                "latency": 14.069,
                "stderr": 1.675,
                "cost_per_test": 0.043303,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "anthropic/claude-3-7-sonnet-20250219-thinking": {
                "accuracy": 59.207,
                "latency": 32.135,
                "stderr": 1.678,
                "cost_per_test": 0.452639,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 18384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001": {
                "accuracy": 58.858,
                "latency": 10.123,
                "stderr": 1.68,
                "cost_per_test": 0.150828,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "fireworks/deepseek-v3p2-thinking": {
                "accuracy": 58.858,
                "latency": 15.178,
                "stderr": 1.68,
                "cost_per_test": 0.0018,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 20480,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "kimi/kimi-k2-thinking": {
                "accuracy": 58.392,
                "latency": 122.216,
                "stderr": 1.683,
                "cost_per_test": 0.06934,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "google/gemini-2.5-pro-exp-03-25": {
                "accuracy": 57.809,
                "latency": 11.897,
                "stderr": 1.686,
                "cost_per_test": 0.169274,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-3": {
                "accuracy": 57.11,
                "latency": 49.7,
                "stderr": 1.69,
                "cost_per_test": 0.317909,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "poolside/laguna-m.1": {
                "accuracy": 56.993,
                "latency": 298.129,
                "stderr": 1.69,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "google/gemini-2.5-pro": {
                "accuracy": 56.876,
                "latency": 20.714,
                "stderr": 1.691,
                "cost_per_test": 0.003627,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "minimax/MiniMax-M2.5": {
                "accuracy": 56.76,
                "latency": 21.903,
                "stderr": 1.691,
                "cost_per_test": 0.045262,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "openai/o3-2025-04-16": {
                "accuracy": 56.061,
                "latency": 28.918,
                "stderr": 1.694,
                "cost_per_test": 0.260856,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-3-mini-fast-high-reasoning": {
                "accuracy": 55.828,
                "latency": 20.701,
                "stderr": 1.695,
                "cost_per_test": 0.032067,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "zai/glm-4.5": {
                "accuracy": 55.478,
                "latency": 136.325,
                "stderr": 1.697,
                "cost_per_test": 0.069957,
                "temperature": 0.6,
                "top_p": 1.0,
                "max_output_tokens": 81920,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "openai/gpt-4.1-mini-2025-04-14": {
                "accuracy": 55.012,
                "latency": 8.221,
                "stderr": 1.698,
                "cost_per_test": 0.051823,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/o4-mini-2025-04-16": {
                "accuracy": 55.012,
                "latency": 27.127,
                "stderr": 1.698,
                "cost_per_test": 0.144774,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "fireworks/deepseek-v3p2": {
                "accuracy": 54.895,
                "latency": 5.336,
                "stderr": 1.699,
                "cost_per_test": 0.000945,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 20480,
                "reasoning": null,
                "reasoning_effort": "none",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "poolside/laguna-xs.2": {
                "accuracy": 54.429,
                "latency": 277.907,
                "stderr": 1.7,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "minimax/MiniMax-M2.1": {
                "accuracy": 54.312,
                "latency": 28.337,
                "stderr": 1.701,
                "cost_per_test": 0.041668,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "grok/grok-4-fast-non-reasoning": {
                "accuracy": 54.079,
                "latency": 13.932,
                "stderr": 1.701,
                "cost_per_test": 0.07579,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-3-5-sonnet-20241022": {
                "accuracy": 53.263,
                "latency": 0.049,
                "stderr": 1.703,
                "cost_per_test": 0.486092,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "together/moonshotai/Kimi-K2-Instruct": {
                "accuracy": 53.03,
                "latency": 33.801,
                "stderr": 1.704,
                "cost_per_test": 0.001714,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "alibaba/qwen3-max": {
                "accuracy": 52.681,
                "latency": 116.245,
                "stderr": 1.705,
                "cost_per_test": 0.177377,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514": {
                "accuracy": 52.681,
                "latency": 449.724,
                "stderr": 1.705,
                "cost_per_test": 0.452376,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-04-17": {
                "accuracy": 52.331,
                "latency": 20.232,
                "stderr": 1.705,
                "cost_per_test": 0.040537,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-3-mini-fast-low-reasoning": {
                "accuracy": 52.098,
                "latency": 14.801,
                "stderr": 1.705,
                "cost_per_test": 0.031819,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "low",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "fireworks/gpt-oss-120b": {
                "accuracy": 51.981,
                "latency": 41.001,
                "stderr": 1.706,
                "cost_per_test": 0.017607,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "fireworks/nemotron-lightning-3p5-30b-a3b": {
                "accuracy": 51.865,
                "latency": 59.671,
                "stderr": 1.706,
                "cost_per_test": 0.009675,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "fireworks/deepseek-v3-0324": {
                "accuracy": 50.932,
                "latency": 66.39,
                "stderr": 1.707,
                "cost_per_test": 0.098456,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "fireworks/deepseek-r1": {
                "accuracy": 50.233,
                "latency": 83.286,
                "stderr": 1.707,
                "cost_per_test": 0.166784,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "arcee-ai/trinity-large-thinking": {
                "accuracy": 49.184,
                "latency": 16.74,
                "stderr": 1.707,
                "cost_per_test": 0.041363,
                "temperature": 0.3,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Arcee-Ai",
                "harness": null
            },
            "zai/glm-4.6": {
                "accuracy": 49.184,
                "latency": 155.354,
                "stderr": 1.707,
                "cost_per_test": 0.089158,
                "temperature": 0.6,
                "top_p": 1.0,
                "max_output_tokens": 122880,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "fireworks/deepseek-v3p1": {
                "accuracy": 48.252,
                "latency": 96.378,
                "stderr": 1.706,
                "cost_per_test": 0.075218,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "fireworks/deepseek-v3": {
                "accuracy": 47.669,
                "latency": 50.565,
                "stderr": 1.705,
                "cost_per_test": 0.098432,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "fireworks/gpt-oss-20b": {
                "accuracy": 46.97,
                "latency": 26.239,
                "stderr": 1.704,
                "cost_per_test": 0.007679,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "fireworks/llama4-maverick-instruct-basic": {
                "accuracy": 45.221,
                "latency": 6.636,
                "stderr": 1.699,
                "cost_per_test": 0.028134,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "grok/grok-2-1212": {
                "accuracy": 45.221,
                "latency": 241.891,
                "stderr": 1.699,
                "cost_per_test": 0.216047,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 0,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-3-5-haiku-20241022": {
                "accuracy": 44.755,
                "latency": 0.009,
                "stderr": 1.698,
                "cost_per_test": 0.129482,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "mistralai/mistral-medium-2505": {
                "accuracy": 44.522,
                "latency": 43.687,
                "stderr": 1.697,
                "cost_per_test": 0.042618,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "alibaba/qwen3-max-preview": {
                "accuracy": 44.056,
                "latency": 441.206,
                "stderr": 1.695,
                "cost_per_test": 0.205935,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "together/meta-llama/Llama-4-Scout-17B-16E-Instruct": {
                "accuracy": 43.124,
                "latency": 20.149,
                "stderr": 1.691,
                "cost_per_test": 0.023152,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "mistralai/magistral-medium-2509": {
                "accuracy": 43.007,
                "latency": 96.844,
                "stderr": 1.69,
                "cost_per_test": 0.216228,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "openai/gpt-4o-2024-11-20": {
                "accuracy": 42.657,
                "latency": 10.92,
                "stderr": 1.688,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "cohere/command-a-03-2025": {
                "accuracy": 41.958,
                "latency": 27.747,
                "stderr": 1.685,
                "cost_per_test": 0.445899,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 8000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "openai/gpt-4o-mini-2024-07-18": {
                "accuracy": 41.026,
                "latency": 0.055,
                "stderr": 1.679,
                "cost_per_test": 0.017023,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-1.5-pro-002": {
                "accuracy": 40.443,
                "latency": 73.047,
                "stderr": 1.675,
                "cost_per_test": 0.229417,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-1.5-flash-002": {
                "accuracy": 38.928,
                "latency": 58.446,
                "stderr": 1.665,
                "cost_per_test": 0.013757,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-2.0-pro-exp-02-05": {
                "accuracy": 38.695,
                "latency": 35.856,
                "stderr": 1.663,
                "cost_per_test": 0.229335,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-4.1-nano-2025-04-14": {
                "accuracy": 38.462,
                "latency": 5.948,
                "stderr": 1.661,
                "cost_per_test": 0.012861,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "mistralai/mistral-small-2503": {
                "accuracy": 38.462,
                "latency": 12.923,
                "stderr": 1.661,
                "cost_per_test": 0.007811,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "mistralai/magistral-small-2509": {
                "accuracy": 36.597,
                "latency": 18.965,
                "stderr": 1.644,
                "cost_per_test": 0.053106,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "openai/gpt-4o-2024-08-06": {
                "accuracy": 36.364,
                "latency": 0.057,
                "stderr": 1.642,
                "cost_per_test": 0.283471,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "ai21labs/jamba-1.5-large": {
                "accuracy": 36.364,
                "latency": 19.187,
                "stderr": 1.642,
                "cost_per_test": 0.304754,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "ai21labs/jamba-large-1.6": {
                "accuracy": 35.431,
                "latency": 40.708,
                "stderr": 1.633,
                "cost_per_test": 0.306016,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "openai/o3-mini-2025-01-31": {
                "accuracy": 33.566,
                "latency": 57.794,
                "stderr": 1.612,
                "cost_per_test": 0.155768,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "ai21labs/jamba-mini-1.6": {
                "accuracy": 32.051,
                "latency": 7.8,
                "stderr": 1.593,
                "cost_per_test": 0.03048,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "together/meta-llama/Meta-Llama-3.1-70B-Instruct-Turbo": {
                "accuracy": 30.653,
                "latency": 0.046,
                "stderr": 1.574,
                "cost_per_test": 0.100898,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "ai21labs/jamba-1.5-mini": {
                "accuracy": 30.07,
                "latency": 4.221,
                "stderr": 1.566,
                "cost_per_test": 0.030431,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "google/gemini-2.0-flash-001": {
                "accuracy": 30.07,
                "latency": 53.247,
                "stderr": 1.566,
                "cost_per_test": 0.018339,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-1.5-flash-001": {
                "accuracy": 25.991,
                "latency": 0.055,
                "stderr": 1.497,
                "cost_per_test": 0.014092,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "together/meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo": {
                "accuracy": 25.641,
                "latency": 0.049,
                "stderr": 1.491,
                "cost_per_test": 0.020359,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "zai/glm-4.7": {
                "accuracy": 17.949,
                "latency": 93.697,
                "stderr": 1.31,
                "cost_per_test": 0.067835,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            }
        },
        "shared_max_context": {
            "anthropic/claude-opus-5": {
                "accuracy": 74.942,
                "latency": 29.037,
                "stderr": 1.479,
                "cost_per_test": 0.696945,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "meta/muse_spark_1_1": {
                "accuracy": 73.077,
                "latency": 25.409,
                "stderr": 1.514,
                "cost_per_test": 0.087456,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "meta/muse_spark_1_2": {
                "accuracy": 72.727,
                "latency": 31.073,
                "stderr": 1.52,
                "cost_per_test": 0.086787,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "kimi/kimi-k3": {
                "accuracy": 72.611,
                "latency": 30.106,
                "stderr": 1.522,
                "cost_per_test": 0.056251,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "anthropic/claude-fable-5": {
                "accuracy": 71.562,
                "latency": 58.898,
                "stderr": 1.54,
                "cost_per_test": 1.387598,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-4-fast-reasoning": {
                "accuracy": 71.096,
                "latency": 10.442,
                "stderr": 1.548,
                "cost_per_test": 0.043356,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "meta/muse_spark": {
                "accuracy": 70.513,
                "latency": 39.26,
                "stderr": 1.557,
                "cost_per_test": 0.010715,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 70.163,
                "latency": 69.331,
                "stderr": 1.562,
                "cost_per_test": 0.043611,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "grok/grok-4.3": {
                "accuracy": 70.047,
                "latency": 24.725,
                "stderr": 1.564,
                "cost_per_test": 0.081293,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "thinkingmachines/inkling-small": {
                "accuracy": 69.814,
                "latency": 51.921,
                "stderr": 1.567,
                "cost_per_test": 0.021752,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 262000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "minimax/MiniMax-M3": {
                "accuracy": 69.697,
                "latency": 23.548,
                "stderr": 1.569,
                "cost_per_test": 0.041574,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 512000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 69.58,
                "latency": 26.764,
                "stderr": 1.571,
                "cost_per_test": 0.362691,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4-0709": {
                "accuracy": 69.347,
                "latency": 33.813,
                "stderr": 1.574,
                "cost_per_test": 0.204668,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "thinkingmachines/inkling": {
                "accuracy": 69.231,
                "latency": 45.083,
                "stderr": 1.576,
                "cost_per_test": 0.072205,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 68.998,
                "latency": 22.376,
                "stderr": 1.579,
                "cost_per_test": 0.478497,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3-max-2026-01-23": {
                "accuracy": 68.531,
                "latency": 88.951,
                "stderr": 1.585,
                "cost_per_test": 0.198727,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "grok/grok-4.5": {
                "accuracy": 68.415,
                "latency": 12.076,
                "stderr": 1.587,
                "cost_per_test": 0.131473,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 68.182,
                "latency": 96.974,
                "stderr": 1.59,
                "cost_per_test": 0.042064,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 48000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "anthropic/claude-sonnet-5": {
                "accuracy": 67.949,
                "latency": 20.457,
                "stderr": 1.593,
                "cost_per_test": 0.416108,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3.6-max-preview": {
                "accuracy": 67.949,
                "latency": 51.146,
                "stderr": 1.593,
                "cost_per_test": 0.102408,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "grok/grok-4.6": {
                "accuracy": 67.832,
                "latency": 24.057,
                "stderr": 1.595,
                "cost_per_test": 0.132183,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "alibaba/qwen3.5-plus-thinking": {
                "accuracy": 67.832,
                "latency": 40.886,
                "stderr": 1.595,
                "cost_per_test": 0.034482,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 67.716,
                "latency": 14.833,
                "stderr": 1.596,
                "cost_per_test": 0.039017,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 67.599,
                "latency": 18.561,
                "stderr": 1.598,
                "cost_per_test": 0.012735,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "alibaba/qwen3.8-max": {
                "accuracy": 67.366,
                "latency": 55.876,
                "stderr": 1.601,
                "cost_per_test": 0.146023,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "zai/glm-5-thinking": {
                "accuracy": 67.016,
                "latency": 70.517,
                "stderr": 1.605,
                "cost_per_test": 0.073316,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "deepseek/deepseek-v4-pro-0813": {
                "accuracy": 66.783,
                "latency": 38.144,
                "stderr": 1.608,
                "cost_per_test": 0.002385,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 66.783,
                "latency": 122.651,
                "stderr": 1.608,
                "cost_per_test": 0.197126,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 66.667,
                "latency": 20.037,
                "stderr": 1.609,
                "cost_per_test": 0.287644,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.6-terra": {
                "accuracy": 66.434,
                "latency": 9.294,
                "stderr": 1.612,
                "cost_per_test": 0.138028,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.6-sol": {
                "accuracy": 66.434,
                "latency": 20.315,
                "stderr": 1.612,
                "cost_per_test": 0.367275,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-5.2": {
                "accuracy": 66.317,
                "latency": 26.348,
                "stderr": 1.614,
                "cost_per_test": 0.0982,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 66.2,
                "latency": 20.361,
                "stderr": 1.615,
                "cost_per_test": 0.475505,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-4-7": {
                "accuracy": 66.084,
                "latency": 18.221,
                "stderr": 1.616,
                "cost_per_test": 0.654716,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-4-8": {
                "accuracy": 65.967,
                "latency": 20.38,
                "stderr": 1.618,
                "cost_per_test": 0.679358,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 65.851,
                "latency": 9.418,
                "stderr": 1.619,
                "cost_per_test": 0.130871,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 65.734,
                "latency": 28.137,
                "stderr": 1.62,
                "cost_per_test": 0.145587,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 65.734,
                "latency": 31.057,
                "stderr": 1.62,
                "cost_per_test": 0.109551,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-4.1-2025-04-14": {
                "accuracy": 65.501,
                "latency": 10.865,
                "stderr": 1.623,
                "cost_per_test": 0.129954,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3.7-max": {
                "accuracy": 65.385,
                "latency": 18.336,
                "stderr": 1.624,
                "cost_per_test": 0.17462,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "zai/glm-5.1": {
                "accuracy": 65.268,
                "latency": 45.736,
                "stderr": 1.625,
                "cost_per_test": 0.071771,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "openai/gpt-5.6-luna": {
                "accuracy": 65.035,
                "latency": 20.103,
                "stderr": 1.628,
                "cost_per_test": 0.01626,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 65.035,
                "latency": 32.888,
                "stderr": 1.628,
                "cost_per_test": 0.090475,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "nvidia/nemotron-3-ultra-550b-a55b": {
                "accuracy": 64.918,
                "latency": 15.701,
                "stderr": 1.629,
                "cost_per_test": null,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Nvidia",
                "harness": null
            },
            "google/gemini-3.5-flash": {
                "accuracy": 64.569,
                "latency": 8.593,
                "stderr": 1.633,
                "cost_per_test": 0.113913,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 64.569,
                "latency": 23.778,
                "stderr": 1.633,
                "cost_per_test": 0.154515,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514-thinking": {
                "accuracy": 64.452,
                "latency": 119.041,
                "stderr": 1.634,
                "cost_per_test": 0.232409,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 64.219,
                "latency": 21.997,
                "stderr": 1.636,
                "cost_per_test": 0.286695,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3.5-flash": {
                "accuracy": 64.219,
                "latency": 34.465,
                "stderr": 1.636,
                "cost_per_test": 0.007994,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "kimi/kimi-k2-thinking": {
                "accuracy": 64.219,
                "latency": 89.86,
                "stderr": 1.636,
                "cost_per_test": 0.041781,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 63.753,
                "latency": 11.623,
                "stderr": 1.641,
                "cost_per_test": 0.023635,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "google/gemini-3.6-flash": {
                "accuracy": 63.52,
                "latency": 6.733,
                "stderr": 1.643,
                "cost_per_test": 0.110554,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-3": {
                "accuracy": 63.52,
                "latency": 17.873,
                "stderr": 1.643,
                "cost_per_test": 0.184299,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 63.17,
                "latency": 15.049,
                "stderr": 1.647,
                "cost_per_test": 0.096327,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "zai/glm-4.6": {
                "accuracy": 63.17,
                "latency": 62.917,
                "stderr": 1.647,
                "cost_per_test": 0.040881,
                "temperature": 0.6,
                "top_p": 1.0,
                "max_output_tokens": 122880,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "zai/glm-4.5": {
                "accuracy": 63.054,
                "latency": 92.297,
                "stderr": 1.648,
                "cost_per_test": 0.040809,
                "temperature": 0.6,
                "top_p": 1.0,
                "max_output_tokens": 81920,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "google/gemini-2.5-pro": {
                "accuracy": 62.587,
                "latency": 28.746,
                "stderr": 1.652,
                "cost_per_test": 0.081072,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "xiaomi/mimo-v2.5-pro": {
                "accuracy": 62.471,
                "latency": 16.623,
                "stderr": 1.653,
                "cost_per_test": 0.029339,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101": {
                "accuracy": 62.354,
                "latency": 10.956,
                "stderr": 1.654,
                "cost_per_test": 0.431934,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3.6-27b": {
                "accuracy": 62.354,
                "latency": 51.777,
                "stderr": 1.654,
                "cost_per_test": 0.046263,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 62.354,
                "latency": 158.5,
                "stderr": 1.654,
                "cost_per_test": 0.079354,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3.5-flash-lite": {
                "accuracy": 62.238,
                "latency": 4.411,
                "stderr": 1.655,
                "cost_per_test": 0.023324,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "ant/ling-3.0-flash-2607": {
                "accuracy": 62.238,
                "latency": 12.856,
                "stderr": 1.655,
                "cost_per_test": 0.005258,
                "temperature": 0.6,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "anthropic/claude-3-7-sonnet-20250219-thinking": {
                "accuracy": 62.238,
                "latency": 23.407,
                "stderr": 1.655,
                "cost_per_test": 0.228911,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 18384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 62.238,
                "latency": 35.919,
                "stderr": 1.655,
                "cost_per_test": 0.038254,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "deepseek/deepseek-v4-flash-0731": {
                "accuracy": 62.005,
                "latency": 17.433,
                "stderr": 1.657,
                "cost_per_test": 0.009434,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025-thinking": {
                "accuracy": 61.888,
                "latency": 30.263,
                "stderr": 1.658,
                "cost_per_test": 0.020614,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 61.888,
                "latency": 74.027,
                "stderr": 1.658,
                "cost_per_test": 0.122195,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001": {
                "accuracy": 61.772,
                "latency": 8.526,
                "stderr": 1.659,
                "cost_per_test": 0.073944,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-2.5-pro-exp-03-25": {
                "accuracy": 61.772,
                "latency": 20.235,
                "stderr": 1.659,
                "cost_per_test": 0.084215,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-3-mini-fast-high-reasoning": {
                "accuracy": 61.655,
                "latency": 14.71,
                "stderr": 1.66,
                "cost_per_test": 0.018712,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "zai/glm-4.7": {
                "accuracy": 61.189,
                "latency": 65.391,
                "stderr": 1.664,
                "cost_per_test": 0.045032,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 61.189,
                "latency": 109.941,
                "stderr": 1.664,
                "cost_per_test": 0.019795,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 61.072,
                "latency": 7.363,
                "stderr": 1.665,
                "cost_per_test": 0.012353,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "minimax/MiniMax-M2.1": {
                "accuracy": 61.072,
                "latency": 22.18,
                "stderr": 1.665,
                "cost_per_test": 0.024085,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 60.839,
                "latency": 4.708,
                "stderr": 1.666,
                "cost_per_test": 0.018404,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "fireworks/gpt-oss-120b": {
                "accuracy": 60.839,
                "latency": 40.371,
                "stderr": 1.666,
                "cost_per_test": 0.010328,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-5-20250929": {
                "accuracy": 60.49,
                "latency": 14.192,
                "stderr": 1.669,
                "cost_per_test": 0.22031,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "minimax/MiniMax-M2.5": {
                "accuracy": 60.373,
                "latency": 17.59,
                "stderr": 1.67,
                "cost_per_test": 0.025373,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025": {
                "accuracy": 60.256,
                "latency": 28.917,
                "stderr": 1.671,
                "cost_per_test": 0.020583,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 60.256,
                "latency": 45.279,
                "stderr": 1.671,
                "cost_per_test": 0.068348,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "xiaomi/mimo-v2.5": {
                "accuracy": 60.14,
                "latency": 19.334,
                "stderr": 1.672,
                "cost_per_test": 0.008752,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "mistralai/mistral-medium-3.5": {
                "accuracy": 59.907,
                "latency": 42.14,
                "stderr": 1.673,
                "cost_per_test": 0.127154,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "openai/o3-2025-04-16": {
                "accuracy": 59.557,
                "latency": 18.054,
                "stderr": 1.675,
                "cost_per_test": 0.131154,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "mistralai/mistral-large-2512": {
                "accuracy": 59.557,
                "latency": 73.135,
                "stderr": 1.675,
                "cost_per_test": 0.035322,
                "temperature": 0.065,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "openai/gpt-4.1-mini-2025-04-14": {
                "accuracy": 59.324,
                "latency": 7.8,
                "stderr": 1.677,
                "cost_per_test": 0.025626,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-3-mini-fast-low-reasoning": {
                "accuracy": 59.324,
                "latency": 11.07,
                "stderr": 1.677,
                "cost_per_test": 0.018464,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "low",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025-thinking": {
                "accuracy": 58.858,
                "latency": 35.154,
                "stderr": 1.68,
                "cost_per_test": 0.006747,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/o4-mini-2025-04-16": {
                "accuracy": 57.925,
                "latency": 17.095,
                "stderr": 1.685,
                "cost_per_test": 0.073049,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "poolside/laguna-m.1": {
                "accuracy": 56.527,
                "latency": 146.262,
                "stderr": 1.692,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "grok/grok-4-fast-non-reasoning": {
                "accuracy": 56.294,
                "latency": 7.342,
                "stderr": 1.693,
                "cost_per_test": 0.042996,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-04-17": {
                "accuracy": 55.361,
                "latency": 5.088,
                "stderr": 1.697,
                "cost_per_test": 0.020182,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3-max": {
                "accuracy": 55.361,
                "latency": 43.689,
                "stderr": 1.697,
                "cost_per_test": 0.080507,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025": {
                "accuracy": 55.245,
                "latency": 34.139,
                "stderr": 1.698,
                "cost_per_test": 0.006824,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "poolside/laguna-xs.2": {
                "accuracy": 55.012,
                "latency": 139.86,
                "stderr": 1.698,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "alibaba/qwen3-max-preview": {
                "accuracy": 54.895,
                "latency": 235.587,
                "stderr": 1.699,
                "cost_per_test": 0.080618,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "arcee-ai/trinity-large-thinking": {
                "accuracy": 54.779,
                "latency": 11.372,
                "stderr": 1.699,
                "cost_per_test": 0.015928,
                "temperature": 0.3,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Arcee-Ai",
                "harness": null
            },
            "fireworks/gpt-oss-20b": {
                "accuracy": 53.497,
                "latency": 43.407,
                "stderr": 1.703,
                "cost_per_test": 0.005289,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "fireworks/deepseek-v3-0324": {
                "accuracy": 52.914,
                "latency": 31.344,
                "stderr": 1.704,
                "cost_per_test": 0.056273,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514": {
                "accuracy": 52.914,
                "latency": 104.11,
                "stderr": 1.704,
                "cost_per_test": 0.221455,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "fireworks/deepseek-r1": {
                "accuracy": 52.564,
                "latency": 36.63,
                "stderr": 1.705,
                "cost_per_test": 0.088122,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "fireworks/nemotron-lightning-3p5-30b-a3b": {
                "accuracy": 52.331,
                "latency": 29.311,
                "stderr": 1.705,
                "cost_per_test": 0.004211,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "anthropic/claude-3-5-haiku-20241022": {
                "accuracy": 51.282,
                "latency": 0.046,
                "stderr": 1.706,
                "cost_per_test": 0.058611,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "fireworks/deepseek-v3": {
                "accuracy": 51.049,
                "latency": 27.684,
                "stderr": 1.707,
                "cost_per_test": 0.05626,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "grok/grok-2-1212": {
                "accuracy": 50.932,
                "latency": 15.85,
                "stderr": 1.707,
                "cost_per_test": 0.129243,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 0,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-3-5-sonnet-20241022": {
                "accuracy": 49.767,
                "latency": 0.052,
                "stderr": 1.707,
                "cost_per_test": 0.219957,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "fireworks/deepseek-v3p1": {
                "accuracy": 49.301,
                "latency": 60.891,
                "stderr": 1.707,
                "cost_per_test": 0.035907,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "fireworks/deepseek-v3p2-thinking": {
                "accuracy": 49.301,
                "latency": 109.73,
                "stderr": 1.707,
                "cost_per_test": 0.037839,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 20480,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "together/moonshotai/Kimi-K2-Instruct": {
                "accuracy": 48.718,
                "latency": 42.666,
                "stderr": 1.706,
                "cost_per_test": 0.064891,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "fireworks/llama4-maverick-instruct-basic": {
                "accuracy": 48.135,
                "latency": 4.369,
                "stderr": 1.706,
                "cost_per_test": 0.013864,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "mistralai/magistral-medium-2509": {
                "accuracy": 47.203,
                "latency": 51.99,
                "stderr": 1.704,
                "cost_per_test": 0.148302,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "grok/grok-4-1-fast-non-reasoning": {
                "accuracy": 46.503,
                "latency": 7.659,
                "stderr": 1.703,
                "cost_per_test": 0.012319,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "mistralai/mistral-medium-2505": {
                "accuracy": 46.154,
                "latency": 35.417,
                "stderr": 1.702,
                "cost_per_test": 0.028549,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "openai/o3-mini-2025-01-31": {
                "accuracy": 45.921,
                "latency": 23.233,
                "stderr": 1.701,
                "cost_per_test": 0.073609,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "fireworks/deepseek-v3p2": {
                "accuracy": 45.455,
                "latency": 40.394,
                "stderr": 1.7,
                "cost_per_test": 0.035821,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 20480,
                "reasoning": null,
                "reasoning_effort": "none",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "together/meta-llama/Llama-4-Scout-17B-16E-Instruct": {
                "accuracy": 44.522,
                "latency": 6.467,
                "stderr": 1.697,
                "cost_per_test": 0.011399,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "openai/gpt-4o-mini-2024-07-18": {
                "accuracy": 43.124,
                "latency": 0.056,
                "stderr": 1.691,
                "cost_per_test": 0.009254,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "mistralai/mistral-small-2503": {
                "accuracy": 41.608,
                "latency": 12.926,
                "stderr": 1.683,
                "cost_per_test": 0.00524,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "google/gemini-2.0-pro-exp-02-05": {
                "accuracy": 40.443,
                "latency": 15.506,
                "stderr": 1.675,
                "cost_per_test": 0.081991,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-1.5-pro-002": {
                "accuracy": 40.443,
                "latency": 20.551,
                "stderr": 1.675,
                "cost_per_test": 0.082015,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "mistralai/magistral-small-2509": {
                "accuracy": 39.977,
                "latency": 14.674,
                "stderr": 1.672,
                "cost_per_test": 0.035917,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "openai/gpt-4o-2024-11-20": {
                "accuracy": 39.86,
                "latency": 6.248,
                "stderr": 1.672,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "cohere/command-a-03-2025": {
                "accuracy": 39.744,
                "latency": 10.862,
                "stderr": 1.671,
                "cost_per_test": 0.166879,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 8000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "ai21labs/jamba-large-1.6": {
                "accuracy": 39.044,
                "latency": 33.916,
                "stderr": 1.665,
                "cost_per_test": 0.139671,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "openai/gpt-4.1-nano-2025-04-14": {
                "accuracy": 38.811,
                "latency": 5.045,
                "stderr": 1.664,
                "cost_per_test": 0.006318,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-1.5-flash-002": {
                "accuracy": 37.879,
                "latency": 13.387,
                "stderr": 1.656,
                "cost_per_test": 0.004915,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "together/meta-llama/Meta-Llama-3.1-70B-Instruct-Turbo": {
                "accuracy": 37.646,
                "latency": 0.068,
                "stderr": 1.654,
                "cost_per_test": 0.057026,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "ai21labs/jamba-mini-1.6": {
                "accuracy": 37.063,
                "latency": 3.391,
                "stderr": 1.649,
                "cost_per_test": 0.013854,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "together/meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo": {
                "accuracy": 36.946,
                "latency": 0.053,
                "stderr": 1.648,
                "cost_per_test": 0.011684,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "ai21labs/jamba-1.5-large": {
                "accuracy": 36.48,
                "latency": 9.97,
                "stderr": 1.643,
                "cost_per_test": 0.138661,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "google/gemini-2.0-flash-001": {
                "accuracy": 35.664,
                "latency": 18.434,
                "stderr": 1.635,
                "cost_per_test": 0.006552,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "ai21labs/jamba-1.5-mini": {
                "accuracy": 31.935,
                "latency": 2.07,
                "stderr": 1.592,
                "cost_per_test": 0.013827,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "openai/gpt-4o-2024-08-06": {
                "accuracy": 31.702,
                "latency": 0.009,
                "stderr": 1.589,
                "cost_per_test": 0.15394,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-1.5-flash-001": {
                "accuracy": 21.445,
                "latency": 0.059,
                "stderr": 1.401,
                "cost_per_test": 0.005006,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            }
        }
    }
}