{
    "metadata": {
        "benchmark": "MMLU Pro",
        "slug": "mmlu_pro",
        "description": "Academic multiple-choice benchmark covering 14 subjects including STEM, humanities, and social sciences.",
        "benchmark_id": "mmlu_pro",
        "family": "mmlu_pro",
        "version": "1",
        "updated": "2026-09-01",
        "dataset_type": "public",
        "industry": "academic",
        "tasks": {
            "overall": "Overall",
            "biology": "Biology",
            "business": "Business",
            "chemistry": "Chemistry",
            "computer_science": "Computer Science",
            "economics": "Economics",
            "engineering": "Engineering",
            "health": "Health",
            "history": "History",
            "law": "Law",
            "math": "Math",
            "other": "Others",
            "philosophy": "Philosophy",
            "physics": "Physics",
            "psychology": "Psychology"
        },
        "models": [
            "ai21labs/jamba-large-1.6",
            "ai21labs/jamba-mini-1.6",
            "alibaba/qwen3-max",
            "alibaba/qwen3-max-2026-01-23",
            "alibaba/qwen3-max-preview",
            "alibaba/qwen3.5-flash",
            "alibaba/qwen3.5-plus-thinking",
            "alibaba/qwen3.6-plus",
            "alibaba/qwen3.7-max",
            "alibaba/qwen3.8-27b",
            "alibaba/qwen3.8-max",
            "ant/ling-3.0-flash-2607",
            "anthropic/claude-3-5-haiku-20241022",
            "anthropic/claude-3-5-sonnet-20241022",
            "anthropic/claude-3-7-sonnet-20250219",
            "anthropic/claude-3-7-sonnet-20250219-thinking",
            "anthropic/claude-fable-5",
            "anthropic/claude-fable-5-1",
            "anthropic/claude-haiku-4-5-20251001-thinking",
            "anthropic/claude-opus-4-1-20250805",
            "anthropic/claude-opus-4-1-20250805-thinking",
            "anthropic/claude-opus-4-20250514",
            "anthropic/claude-opus-4-5-20251101",
            "anthropic/claude-opus-4-5-20251101-thinking",
            "anthropic/claude-opus-4-6-thinking",
            "anthropic/claude-opus-4-7",
            "anthropic/claude-opus-4-8",
            "anthropic/claude-opus-5",
            "anthropic/claude-sonnet-4-20250514",
            "anthropic/claude-sonnet-4-20250514-thinking",
            "anthropic/claude-sonnet-4-5-20250929-thinking",
            "anthropic/claude-sonnet-4-6",
            "anthropic/claude-sonnet-5",
            "cohere/command-a-03-2025",
            "cohere/command-r-plus",
            "deepseek/deepseek-v4-flash-0731",
            "deepseek/deepseek-v4-pro",
            "deepseek/deepseek-v4-pro-0813",
            "fireworks/deepseek-r1",
            "fireworks/deepseek-v3",
            "fireworks/deepseek-v3-0324",
            "fireworks/deepseek-v3p2",
            "fireworks/deepseek-v3p2-thinking",
            "fireworks/gpt-oss-120b",
            "fireworks/gpt-oss-20b",
            "fireworks/llama4-maverick-instruct-basic",
            "fireworks/qwen3-235b-a22b",
            "google/gemini-1.5-flash-002",
            "google/gemini-1.5-pro-002",
            "google/gemini-2.0-flash-001",
            "google/gemini-2.5-flash-lite-preview-09-2025",
            "google/gemini-2.5-flash-lite-preview-09-2025-thinking",
            "google/gemini-2.5-flash-preview-09-2025",
            "google/gemini-2.5-flash-preview-09-2025-thinking",
            "google/gemini-2.5-pro-exp-03-25",
            "google/gemini-3-flash-preview",
            "google/gemini-3-pro-preview",
            "google/gemini-3.1-flash-lite-preview",
            "google/gemini-3.1-pro-preview",
            "google/gemini-3.5-flash",
            "google/gemini-3.5-flash-lite",
            "google/gemini-3.6-flash",
            "google/gemini-3.7-flash",
            "google/gemini-3.8-flash",
            "grok/grok-2-1212",
            "grok/grok-3",
            "grok/grok-3-mini-fast-high-reasoning",
            "grok/grok-3-mini-fast-low-reasoning",
            "grok/grok-4-0709",
            "grok/grok-4-1-fast-non-reasoning",
            "grok/grok-4-1-fast-reasoning",
            "grok/grok-4-fast-non-reasoning",
            "grok/grok-4-fast-reasoning",
            "grok/grok-4.20-0309-reasoning",
            "grok/grok-4.3",
            "grok/grok-4.5",
            "grok/grok-4.6",
            "kimi/kimi-k2-thinking",
            "kimi/kimi-k2.5-thinking",
            "kimi/kimi-k2.6",
            "kimi/kimi-k3",
            "meta/muse_spark",
            "meta/muse_spark_1_1",
            "meta/muse_spark_1_2",
            "minimax/MiniMax-M2.1",
            "minimax/MiniMax-M2.5",
            "minimax/MiniMax-M2.7",
            "minimax/MiniMax-M3",
            "mistralai/magistral-medium-2509",
            "mistralai/magistral-small-2509",
            "mistralai/mistral-large-2411",
            "mistralai/mistral-large-2512",
            "mistralai/mistral-medium-2505",
            "mistralai/mistral-medium-3.5",
            "mistralai/mistral-small-2402",
            "mistralai/mistral-small-2503",
            "nvidia/nemotron-3-ultra-550b-a55b",
            "openai/gpt-4.1-2025-04-14",
            "openai/gpt-4.1-mini-2025-04-14",
            "openai/gpt-4.1-nano-2025-04-14",
            "openai/gpt-4o-2024-08-06",
            "openai/gpt-4o-2024-11-20",
            "openai/gpt-4o-mini-2024-07-18",
            "openai/gpt-5-2025-08-07",
            "openai/gpt-5-mini-2025-08-07",
            "openai/gpt-5-nano-2025-08-07",
            "openai/gpt-5.1-2025-11-13",
            "openai/gpt-5.2-2025-12-11",
            "openai/gpt-5.4-2026-03-05",
            "openai/gpt-5.4-mini-2026-03-17",
            "openai/gpt-5.4-nano-2026-03-17",
            "openai/gpt-5.5",
            "openai/gpt-5.6-luna",
            "openai/gpt-5.6-sol",
            "openai/gpt-5.6-terra",
            "openai/o1-2024-12-17",
            "openai/o3-2025-04-16",
            "openai/o3-mini-2025-01-31",
            "openai/o4-mini-2025-04-16",
            "poolside/laguna-m.1",
            "poolside/laguna-xs.2",
            "thinkingmachines/inkling",
            "thinkingmachines/inkling-small",
            "together/langston/nim/nvidia/llama-3.3-nemotron-super-49b-v1-42e84561",
            "together/langston/nim/nvidia/llama-3.3-nemotron-super-49b-v1-42e84561-thinking",
            "together/meta-llama/Llama-3.3-70B-Instruct-Turbo",
            "together/meta-llama/Llama-4-Scout-17B-16E-Instruct",
            "together/moonshotai/Kimi-K2-Instruct",
            "xiaomi/mimo-v2.5",
            "xiaomi/mimo-v2.5-pro",
            "zai/glm-4.5",
            "zai/glm-4.6",
            "zai/glm-4.7",
            "zai/glm-5-thinking",
            "zai/glm-5.1",
            "zai/glm-5.2",
            "zai/glm-5.3",
            "zai/glm-5.3-flash"
        ],
        "partners": [],
        "showBadge": false,
        "visible": true,
        "use_cost_per_test": false,
        "runner": "platform",
        "mode": "one-shot",
        "archived": false,
        "partner": false,
        "total_models": 138
    },
    "tasks": {
        "overall": {
            "anthropic/claude-fable-5-1": {
                "accuracy": 92.375,
                "latency": 20.589,
                "stderr": 0.266,
                "cost_per_test": 0.097724,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-5": {
                "accuracy": 91.591,
                "latency": 9.6,
                "stderr": 0.276,
                "cost_per_test": 0.027843,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-fable-5": {
                "accuracy": 91.502,
                "latency": 24.998,
                "stderr": 0.278,
                "cost_per_test": 0.082276,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 90.987,
                "latency": 23.912,
                "stderr": 0.284,
                "cost_per_test": 0.027545,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.8-flash": {
                "accuracy": 90.221,
                "latency": 7.82,
                "stderr": 0.295,
                "cost_per_test": 0.022457,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.7-flash": {
                "accuracy": 90.121,
                "latency": 4.159,
                "stderr": 0.296,
                "cost_per_test": 0.010817,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 90.102,
                "latency": 24.272,
                "stderr": 0.295,
                "cost_per_test": 0.036926,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-4-7": {
                "accuracy": 89.871,
                "latency": 187.83,
                "stderr": 0.305,
                "cost_per_test": 0.061711,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-4-8": {
                "accuracy": 89.585,
                "latency": 23.44,
                "stderr": 0.303,
                "cost_per_test": 0.060716,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.5-flash": {
                "accuracy": 89.515,
                "latency": 26.262,
                "stderr": 0.306,
                "cost_per_test": 0.02498,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-4.6": {
                "accuracy": 89.4,
                "latency": 54.975,
                "stderr": 0.305,
                "cost_per_test": 0.015391,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "alibaba/qwen3.7-max": {
                "accuracy": 89.311,
                "latency": 29.15,
                "stderr": 0.305,
                "cost_per_test": 0.01678,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "google/gemini-3.6-flash": {
                "accuracy": 89.276,
                "latency": 7.298,
                "stderr": 0.305,
                "cost_per_test": 0.013327,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-4.5": {
                "accuracy": 89.216,
                "latency": 357.751,
                "stderr": 0.31,
                "cost_per_test": 0.014974,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 89.107,
                "latency": 49.451,
                "stderr": 0.455,
                "cost_per_test": 0.052129,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.6-sol": {
                "accuracy": 89.1,
                "latency": 15.654,
                "stderr": 0.308,
                "cost_per_test": 0.032087,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "meta/muse_spark_1_1": {
                "accuracy": 88.726,
                "latency": 21.03,
                "stderr": 0.318,
                "cost_per_test": 0.010704,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "alibaba/qwen3.8-max": {
                "accuracy": 88.602,
                "latency": 63.576,
                "stderr": 0.313,
                "cost_per_test": 0.018585,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 88.592,
                "latency": 36.249,
                "stderr": 0.315,
                "cost_per_test": 0.019364,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "meta/muse_spark_1_2": {
                "accuracy": 88.275,
                "latency": 44.234,
                "stderr": 0.318,
                "cost_per_test": 0.010668,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 88.144,
                "latency": 42.114,
                "stderr": 0.318,
                "cost_per_test": 0.027891,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "kimi/kimi-k3": {
                "accuracy": 87.973,
                "latency": 38.718,
                "stderr": 0.321,
                "cost_per_test": 0.021327,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "anthropic/claude-opus-4-1-20250805-thinking": {
                "accuracy": 87.924,
                "latency": 27.668,
                "stderr": 0.325,
                "cost_per_test": 0.124083,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 87.668,
                "latency": 46.184,
                "stderr": 0.323,
                "cost_per_test": 0.008813,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 87.572,
                "latency": 156.347,
                "stderr": 0.33,
                "cost_per_test": 0.018214,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "anthropic/claude-sonnet-5": {
                "accuracy": 87.545,
                "latency": 25.191,
                "stderr": 0.368,
                "cost_per_test": 0.042996,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 87.482,
                "latency": 28.875,
                "stderr": 0.421,
                "cost_per_test": 0.050888,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 87.357,
                "latency": 38.2,
                "stderr": 0.391,
                "cost_per_test": 0.040152,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 87.341,
                "latency": 55.406,
                "stderr": 0.434,
                "cost_per_test": 0.066412,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "meta/muse_spark": {
                "accuracy": 87.316,
                "latency": 46.242,
                "stderr": 0.327,
                "cost_per_test": 0.00167,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 87.26,
                "latency": 38.92,
                "stderr": 0.381,
                "cost_per_test": 0.215138,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 87.249,
                "latency": 577.617,
                "stderr": 0.335,
                "cost_per_test": 0.027032,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "anthropic/claude-opus-4-1-20250805": {
                "accuracy": 87.214,
                "latency": 17.757,
                "stderr": 0.328,
                "cost_per_test": 0.007078,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3.5-plus-thinking": {
                "accuracy": 87.178,
                "latency": 76.678,
                "stderr": 0.328,
                "cost_per_test": 0.008912,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "minimax/MiniMax-M2.1": {
                "accuracy": 87.049,
                "latency": 40.957,
                "stderr": 0.333,
                "cost_per_test": 0.003225,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "deepseek/deepseek-v4-pro-0813": {
                "accuracy": 86.972,
                "latency": 59.624,
                "stderr": 0.341,
                "cost_per_test": 0.01283,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "zai/glm-5.1": {
                "accuracy": 86.9,
                "latency": 69.318,
                "stderr": 0.334,
                "cost_per_test": 0.012383,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "zai/glm-5.3": {
                "accuracy": 86.77,
                "latency": 61.783,
                "stderr": 0.335,
                "cost_per_test": 0.016639,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "zai/glm-5.2": {
                "accuracy": 86.714,
                "latency": 51.64,
                "stderr": 0.334,
                "cost_per_test": 0.013838,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "openai/gpt-5.6-terra": {
                "accuracy": 86.659,
                "latency": 8.173,
                "stderr": 0.333,
                "cost_per_test": 0.008039,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 86.544,
                "latency": 37.239,
                "stderr": 0.336,
                "cost_per_test": 0.024135,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 86.377,
                "latency": 23.056,
                "stderr": 0.337,
                "cost_per_test": 0.018514,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "thinkingmachines/inkling": {
                "accuracy": 86.295,
                "latency": 73.264,
                "stderr": 0.34,
                "cost_per_test": 0.039629,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 86.254,
                "latency": 11.654,
                "stderr": 0.34,
                "cost_per_test": 0.012836,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 86.24,
                "latency": 14.511,
                "stderr": 0.34,
                "cost_per_test": 0.001052,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 86.234,
                "latency": 35.413,
                "stderr": 0.344,
                "cost_per_test": 0.03698,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "deepseek/deepseek-v4-flash-0731": {
                "accuracy": 86.206,
                "latency": 17.188,
                "stderr": 0.339,
                "cost_per_test": 0.00265,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "anthropic/claude-opus-4-20250514": {
                "accuracy": 86.166,
                "latency": 10.405,
                "stderr": 0.337,
                "cost_per_test": 0.049652,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "zai/glm-5.3-flash": {
                "accuracy": 86.059,
                "latency": 44.762,
                "stderr": 0.341,
                "cost_per_test": 0.001058,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "openai/gpt-5.6-luna": {
                "accuracy": 86.036,
                "latency": 16.948,
                "stderr": 0.35,
                "cost_per_test": 0.002842,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-5-thinking": {
                "accuracy": 86.03,
                "latency": 86.889,
                "stderr": 0.343,
                "cost_per_test": 0.011542,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 85.914,
                "latency": 39.204,
                "stderr": 0.343,
                "cost_per_test": 0.00793,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "grok/grok-4.3": {
                "accuracy": 85.838,
                "latency": 637.088,
                "stderr": 0.33,
                "cost_per_test": 0.009489,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-3.5-flash-lite": {
                "accuracy": 85.836,
                "latency": 4.823,
                "stderr": 0.344,
                "cost_per_test": 0.004522,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "nvidia/nemotron-3-ultra-550b-a55b": {
                "accuracy": 85.761,
                "latency": 32.92,
                "stderr": 0.345,
                "cost_per_test": null,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Nvidia",
                "harness": null
            },
            "openai/o3-2025-04-16": {
                "accuracy": 85.595,
                "latency": 16.735,
                "stderr": 0.345,
                "cost_per_test": 0.011827,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101": {
                "accuracy": 85.59,
                "latency": 8.248,
                "stderr": 0.367,
                "cost_per_test": 0.052065,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "thinkingmachines/inkling-small": {
                "accuracy": 85.566,
                "latency": 71.685,
                "stderr": 0.346,
                "cost_per_test": 0.005706,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 262000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "grok/grok-4-0709": {
                "accuracy": 85.304,
                "latency": 89.255,
                "stderr": 0.349,
                "cost_per_test": 0.05025,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "alibaba/qwen3-max-2026-01-23": {
                "accuracy": 84.976,
                "latency": 110.207,
                "stderr": 0.352,
                "cost_per_test": 0.026557,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "fireworks/deepseek-v3p2-thinking": {
                "accuracy": 84.917,
                "latency": 90.148,
                "stderr": 0.353,
                "cost_per_test": 0.001102,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "xiaomi/mimo-v2.5-pro": {
                "accuracy": 84.595,
                "latency": 67.757,
                "stderr": 0.453,
                "cost_per_test": 0.002085,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 84.554,
                "latency": 20.001,
                "stderr": 0.359,
                "cost_per_test": 0.00368,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3-max": {
                "accuracy": 84.36,
                "latency": 0.0,
                "stderr": 0.356,
                "cost_per_test": 0.006426,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "alibaba/qwen3.8-27b": {
                "accuracy": 84.337,
                "latency": 69.848,
                "stderr": 0.356,
                "cost_per_test": 0.010543,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "minimax/MiniMax-M3": {
                "accuracy": 84.22,
                "latency": 41.39,
                "stderr": 0.364,
                "cost_per_test": 0.004811,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 512000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 84.18,
                "latency": 25.469,
                "stderr": 0.357,
                "cost_per_test": 0.001078,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "alibaba/qwen3.5-flash": {
                "accuracy": 84.061,
                "latency": 33.303,
                "stderr": 0.36,
                "cost_per_test": 0.001734,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "google/gemini-2.5-pro-exp-03-25": {
                "accuracy": 84.058,
                "latency": 18.46,
                "stderr": 0.358,
                "cost_per_test": 0.009006,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514-thinking": {
                "accuracy": 83.865,
                "latency": 47.087,
                "stderr": 0.361,
                "cost_per_test": 0.045537,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025": {
                "accuracy": 83.686,
                "latency": 8.525,
                "stderr": 0.362,
                "cost_per_test": 0.001454,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025-thinking": {
                "accuracy": 83.664,
                "latency": 10.445,
                "stderr": 0.361,
                "cost_per_test": 0.00146,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3-max-preview": {
                "accuracy": 83.537,
                "latency": 43.428,
                "stderr": 0.364,
                "cost_per_test": 0.006194,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/o1-2024-12-17": {
                "accuracy": 83.488,
                "latency": 26.867,
                "stderr": 0.364,
                "cost_per_test": 0.116223,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "fireworks/deepseek-r1": {
                "accuracy": 83.184,
                "latency": 27.278,
                "stderr": 0.444,
                "cost_per_test": 0.012339,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "fireworks/deepseek-v3p2": {
                "accuracy": 83.056,
                "latency": 40.772,
                "stderr": 0.381,
                "cost_per_test": 0.000283,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "xiaomi/mimo-v2.5": {
                "accuracy": 82.931,
                "latency": 30.248,
                "stderr": 0.372,
                "cost_per_test": 0.000774,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "zai/glm-4.7": {
                "accuracy": 82.736,
                "latency": 104.451,
                "stderr": 0.371,
                "cost_per_test": 0.00798,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "anthropic/claude-3-7-sonnet-20250219-thinking": {
                "accuracy": 82.728,
                "latency": 31.744,
                "stderr": 0.37,
                "cost_per_test": 0.040182,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 82.226,
                "latency": 22.196,
                "stderr": 0.373,
                "cost_per_test": 0.005196,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-4.6": {
                "accuracy": 82.203,
                "latency": 46.996,
                "stderr": 0.375,
                "cost_per_test": 0.006572,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 122880,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "ant/ling-3.0-flash-2607": {
                "accuracy": 82.008,
                "latency": 20.073,
                "stderr": 0.375,
                "cost_per_test": 0.000862,
                "temperature": 0.6,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "grok/grok-3-mini-fast-high-reasoning": {
                "accuracy": 81.374,
                "latency": 16.888,
                "stderr": 0.382,
                "cost_per_test": 0.001758,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "fireworks/qwen3-235b-a22b": {
                "accuracy": 81.246,
                "latency": 52.354,
                "stderr": 0.387,
                "cost_per_test": 0.002316,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "zai/glm-4.5": {
                "accuracy": 81.222,
                "latency": 136.72,
                "stderr": 0.378,
                "cost_per_test": 0.009757,
                "temperature": 0.6,
                "top_p": 1.0,
                "max_output_tokens": 81920,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "kimi/kimi-k2-thinking": {
                "accuracy": 81.068,
                "latency": 94.771,
                "stderr": 0.396,
                "cost_per_test": 0.002067,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "anthropic/claude-3-7-sonnet-20250219": {
                "accuracy": 80.663,
                "latency": 6.302,
                "stderr": 0.383,
                "cost_per_test": 0.009664,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/o4-mini-2025-04-16": {
                "accuracy": 80.561,
                "latency": 10.443,
                "stderr": 0.386,
                "cost_per_test": 0.005524,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-4.1-2025-04-14": {
                "accuracy": 80.495,
                "latency": 8.177,
                "stderr": 0.385,
                "cost_per_test": 0.007141,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 80.427,
                "latency": 50.005,
                "stderr": 0.39,
                "cost_per_test": 0.003947,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "minimax/MiniMax-M2.5": {
                "accuracy": 80.093,
                "latency": 48.271,
                "stderr": 0.39,
                "cost_per_test": 0.005362,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "grok/grok-3-mini-fast-low-reasoning": {
                "accuracy": 80.011,
                "latency": 7.426,
                "stderr": 0.39,
                "cost_per_test": 0.001026,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "low",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "grok/grok-3": {
                "accuracy": 79.949,
                "latency": 12.096,
                "stderr": 0.39,
                "cost_per_test": 0.016622,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "mistralai/mistral-large-2512": {
                "accuracy": 79.823,
                "latency": 13.554,
                "stderr": 0.472,
                "cost_per_test": 0.001481,
                "temperature": 0.065,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "grok/grok-4-fast-reasoning": {
                "accuracy": 79.704,
                "latency": 7.343,
                "stderr": 0.393,
                "cost_per_test": 0.001742,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "fireworks/deepseek-v3-0324": {
                "accuracy": 79.474,
                "latency": 25.051,
                "stderr": 0.396,
                "cost_per_test": 0.00147,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514": {
                "accuracy": 79.432,
                "latency": 9.707,
                "stderr": 0.372,
                "cost_per_test": 0.010229,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "fireworks/llama4-maverick-instruct-basic": {
                "accuracy": 79.415,
                "latency": 6.803,
                "stderr": 0.395,
                "cost_per_test": 0.000686,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "together/moonshotai/Kimi-K2-Instruct": {
                "accuracy": 79.394,
                "latency": 20.338,
                "stderr": 0.394,
                "cost_per_test": 0.002836,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "fireworks/gpt-oss-120b": {
                "accuracy": 79.166,
                "latency": 35.41,
                "stderr": 0.395,
                "cost_per_test": 0.001438,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025-thinking": {
                "accuracy": 79.118,
                "latency": 9.618,
                "stderr": 0.397,
                "cost_per_test": 0.000314,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 78.715,
                "latency": 25.784,
                "stderr": 0.477,
                "cost_per_test": 0.018312,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/o3-mini-2025-01-31": {
                "accuracy": 78.689,
                "latency": 22.375,
                "stderr": 0.394,
                "cost_per_test": 0.012991,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025": {
                "accuracy": 78.639,
                "latency": 3.19,
                "stderr": 0.4,
                "cost_per_test": 0.000556,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-3-5-sonnet-20241022": {
                "accuracy": 78.404,
                "latency": 6.288,
                "stderr": 0.397,
                "cost_per_test": 0.00859,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-2.0-flash-001": {
                "accuracy": 77.376,
                "latency": 4.318,
                "stderr": 0.407,
                "cost_per_test": 0.000414,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-4.1-mini-2025-04-14": {
                "accuracy": 77.225,
                "latency": 3.832,
                "stderr": 0.405,
                "cost_per_test": 0.001316,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 77.172,
                "latency": 8.21,
                "stderr": 0.433,
                "cost_per_test": 0.000396,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5-nano-2025-08-07": {
                "accuracy": 76.067,
                "latency": 26.107,
                "stderr": 0.412,
                "cost_per_test": 0.001887,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-2-1212": {
                "accuracy": 75.473,
                "latency": 8.747,
                "stderr": 0.418,
                "cost_per_test": 0.005966,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 0,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "mistralai/mistral-medium-3.5": {
                "accuracy": 75.335,
                "latency": 66.384,
                "stderr": 0.418,
                "cost_per_test": 0.041763,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "google/gemini-1.5-pro-002": {
                "accuracy": 75.294,
                "latency": 3.343,
                "stderr": 0.417,
                "cost_per_test": 0.002649,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "mistralai/mistral-medium-2505": {
                "accuracy": 75.29,
                "latency": 8.831,
                "stderr": 0.421,
                "cost_per_test": 0.000209,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "grok/grok-4-1-fast-non-reasoning": {
                "accuracy": 75.214,
                "latency": 3.098,
                "stderr": 0.419,
                "cost_per_test": 0.000217,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-4o-2024-08-06": {
                "accuracy": 74.13,
                "latency": 8.998,
                "stderr": 0.42,
                "cost_per_test": 0.007217,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "fireworks/deepseek-v3": {
                "accuracy": 73.82,
                "latency": 11.36,
                "stderr": 0.424,
                "cost_per_test": 0.001338,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "openai/gpt-4o-2024-11-20": {
                "accuracy": 72.563,
                "latency": 9.143,
                "stderr": 0.441,
                "cost_per_test": 0.007465,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "fireworks/gpt-oss-20b": {
                "accuracy": 71.636,
                "latency": 30.207,
                "stderr": 0.447,
                "cost_per_test": 0.00062,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "together/langston/nim/nvidia/llama-3.3-nemotron-super-49b-v1-42e84561": {
                "accuracy": 70.778,
                "latency": 17.45,
                "stderr": 0.45,
                "cost_per_test": null,
                "temperature": 0.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "grok/grok-4-fast-non-reasoning": {
                "accuracy": 70.342,
                "latency": 2.135,
                "stderr": 0.431,
                "cost_per_test": 0.001084,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "together/meta-llama/Llama-3.3-70B-Instruct-Turbo": {
                "accuracy": 69.86,
                "latency": 4.179,
                "stderr": 0.442,
                "cost_per_test": 0.001565,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "mistralai/mistral-large-2411": {
                "accuracy": 69.706,
                "latency": 7.192,
                "stderr": 0.442,
                "cost_per_test": 0.004997,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "together/meta-llama/Llama-4-Scout-17B-16E-Instruct": {
                "accuracy": 69.632,
                "latency": 5.342,
                "stderr": 0.445,
                "cost_per_test": 0.000507,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "together/langston/nim/nvidia/llama-3.3-nemotron-super-49b-v1-42e84561-thinking": {
                "accuracy": 69.577,
                "latency": 36.351,
                "stderr": 0.441,
                "cost_per_test": null,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "cohere/command-a-03-2025": {
                "accuracy": 69.172,
                "latency": 9.853,
                "stderr": 0.463,
                "cost_per_test": 0.007902,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 8000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "poolside/laguna-xs.2": {
                "accuracy": 69.05,
                "latency": 47.986,
                "stderr": 0.48,
                "cost_per_test": 0.000605,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "poolside/laguna-m.1": {
                "accuracy": 68.84,
                "latency": 99.583,
                "stderr": 0.452,
                "cost_per_test": 0.001794,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "mistralai/magistral-medium-2509": {
                "accuracy": 68.659,
                "latency": 31.988,
                "stderr": 0.433,
                "cost_per_test": 0.014664,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "mistralai/mistral-small-2503": {
                "accuracy": 66.021,
                "latency": 3.602,
                "stderr": 0.455,
                "cost_per_test": 0.000203,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "google/gemini-1.5-flash-002": {
                "accuracy": 65.606,
                "latency": 1.682,
                "stderr": 0.456,
                "cost_per_test": 0.00017,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "mistralai/mistral-small-2402": {
                "accuracy": 64.443,
                "latency": 4.665,
                "stderr": 0.461,
                "cost_per_test": 0.000502,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "anthropic/claude-3-5-haiku-20241022": {
                "accuracy": 64.117,
                "latency": 5.792,
                "stderr": 0.46,
                "cost_per_test": 0.002224,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-4.1-nano-2025-04-14": {
                "accuracy": 63.479,
                "latency": 2.403,
                "stderr": 0.462,
                "cost_per_test": 0.000337,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-4o-mini-2024-07-18": {
                "accuracy": 62.735,
                "latency": 5.086,
                "stderr": 0.461,
                "cost_per_test": 0.000412,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "mistralai/magistral-small-2509": {
                "accuracy": 62.135,
                "latency": 16.124,
                "stderr": 0.452,
                "cost_per_test": 0.00355,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "ai21labs/jamba-large-1.6": {
                "accuracy": 49.778,
                "latency": 9.48,
                "stderr": 0.475,
                "cost_per_test": 0.004628,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "cohere/command-r-plus": {
                "accuracy": 44.003,
                "latency": 4.419,
                "stderr": 0.479,
                "cost_per_test": 0.00566,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 4000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "ai21labs/jamba-mini-1.6": {
                "accuracy": 30.278,
                "latency": 2.426,
                "stderr": 0.448,
                "cost_per_test": 0.000391,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            }
        },
        "biology": {
            "anthropic/claude-fable-5-1": {
                "accuracy": 96.095,
                "latency": 11.203,
                "stderr": 0.723,
                "cost_per_test": 0.058467,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-5": {
                "accuracy": 95.676,
                "latency": 7.056,
                "stderr": 0.76,
                "cost_per_test": 0.022565,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 95.397,
                "latency": 17.19,
                "stderr": 0.783,
                "cost_per_test": 0.026082,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 95.258,
                "latency": 15.298,
                "stderr": 0.794,
                "cost_per_test": 0.016695,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3.7-max": {
                "accuracy": 95.119,
                "latency": 21.48,
                "stderr": 0.805,
                "cost_per_test": 0.012963,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "google/gemini-3.8-flash": {
                "accuracy": 94.979,
                "latency": 5.07,
                "stderr": 0.816,
                "cost_per_test": 0.013544,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "meta/muse_spark_1_1": {
                "accuracy": 94.979,
                "latency": 15.731,
                "stderr": 0.896,
                "cost_per_test": 0.008217,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "grok/grok-4.6": {
                "accuracy": 94.979,
                "latency": 26.674,
                "stderr": 0.816,
                "cost_per_test": 0.009049,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-3.7-flash": {
                "accuracy": 94.7,
                "latency": 3.04,
                "stderr": 0.837,
                "cost_per_test": 0.007503,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.6-flash": {
                "accuracy": 94.7,
                "latency": 4.901,
                "stderr": 0.837,
                "cost_per_test": 0.009068,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-4-8": {
                "accuracy": 94.7,
                "latency": 16.602,
                "stderr": 0.837,
                "cost_per_test": 0.042,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 94.7,
                "latency": 25.614,
                "stderr": 0.837,
                "cost_per_test": 0.013204,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.6-sol": {
                "accuracy": 94.561,
                "latency": 7.724,
                "stderr": 0.847,
                "cost_per_test": 0.018769,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-4-7": {
                "accuracy": 94.561,
                "latency": 64.371,
                "stderr": 0.847,
                "cost_per_test": 0.03085,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "thinkingmachines/inkling": {
                "accuracy": 94.282,
                "latency": 34.732,
                "stderr": 0.867,
                "cost_per_test": 0.017451,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "zai/glm-5-thinking": {
                "accuracy": 94.282,
                "latency": 57.433,
                "stderr": 0.867,
                "cost_per_test": 0.007637,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "grok/grok-4.5": {
                "accuracy": 94.282,
                "latency": 377.984,
                "stderr": 0.867,
                "cost_per_test": 0.009436,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 94.142,
                "latency": 8.357,
                "stderr": 0.877,
                "cost_per_test": 0.009299,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "alibaba/qwen3.8-max": {
                "accuracy": 94.142,
                "latency": 26.173,
                "stderr": 0.877,
                "cost_per_test": 0.007722,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "alibaba/qwen3.5-plus-thinking": {
                "accuracy": 94.142,
                "latency": 55.706,
                "stderr": 0.877,
                "cost_per_test": 0.006883,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101": {
                "accuracy": 94.003,
                "latency": 8.647,
                "stderr": 0.994,
                "cost_per_test": 0.05251,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 94.003,
                "latency": 20.744,
                "stderr": 1.843,
                "cost_per_test": 0.022805,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 94.003,
                "latency": 24.475,
                "stderr": 0.887,
                "cost_per_test": 0.018287,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-5.2": {
                "accuracy": 94.003,
                "latency": 34.071,
                "stderr": 0.887,
                "cost_per_test": 0.007903,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "google/gemini-3.5-flash": {
                "accuracy": 93.863,
                "latency": 12.325,
                "stderr": 0.896,
                "cost_per_test": 0.015872,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "kimi/kimi-k3": {
                "accuracy": 93.863,
                "latency": 15.082,
                "stderr": 0.896,
                "cost_per_test": 0.0126,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "zai/glm-5.3-flash": {
                "accuracy": 93.863,
                "latency": 19.649,
                "stderr": 0.915,
                "cost_per_test": 0.000484,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "meta/muse_spark_1_2": {
                "accuracy": 93.863,
                "latency": 28.191,
                "stderr": 0.924,
                "cost_per_test": 0.006943,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 93.724,
                "latency": 24.916,
                "stderr": 1.122,
                "cost_per_test": 0.139523,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "deepseek/deepseek-v4-pro-0813": {
                "accuracy": 93.724,
                "latency": 28.033,
                "stderr": 0.924,
                "cost_per_test": 0.00547,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "zai/glm-5.1": {
                "accuracy": 93.724,
                "latency": 29.038,
                "stderr": 0.906,
                "cost_per_test": 0.005623,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "meta/muse_spark": {
                "accuracy": 93.724,
                "latency": 34.668,
                "stderr": 0.906,
                "cost_per_test": 0.001406,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 93.724,
                "latency": 35.767,
                "stderr": 0.906,
                "cost_per_test": 0.007006,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 93.724,
                "latency": 96.93,
                "stderr": 0.906,
                "cost_per_test": 0.010782,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "anthropic/claude-fable-5": {
                "accuracy": 93.584,
                "latency": 18.862,
                "stderr": 0.915,
                "cost_per_test": 0.086148,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "minimax/MiniMax-M2.1": {
                "accuracy": 93.584,
                "latency": 27.098,
                "stderr": 0.924,
                "cost_per_test": 0.0022,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "anthropic/claude-sonnet-5": {
                "accuracy": 93.584,
                "latency": 33.745,
                "stderr": 1.443,
                "cost_per_test": 0.05724,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.6-luna": {
                "accuracy": 93.305,
                "latency": 8.764,
                "stderr": 0.96,
                "cost_per_test": 0.001241,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-5.3": {
                "accuracy": 93.305,
                "latency": 32.613,
                "stderr": 0.933,
                "cost_per_test": 0.007849,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "thinkingmachines/inkling-small": {
                "accuracy": 93.305,
                "latency": 40.714,
                "stderr": 0.933,
                "cost_per_test": 0.002762,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 262000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "grok/grok-4.3": {
                "accuracy": 93.305,
                "latency": 314.075,
                "stderr": 0.933,
                "cost_per_test": 0.00588,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 93.166,
                "latency": 21.666,
                "stderr": 0.942,
                "cost_per_test": 0.000816,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.6-terra": {
                "accuracy": 93.026,
                "latency": 9.097,
                "stderr": 0.951,
                "cost_per_test": 0.00555,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-4-20250514": {
                "accuracy": 93.026,
                "latency": 10.237,
                "stderr": 0.951,
                "cost_per_test": 0.048629,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 93.026,
                "latency": 13.658,
                "stderr": 0.951,
                "cost_per_test": 0.00067,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 93.026,
                "latency": 24.712,
                "stderr": 0.951,
                "cost_per_test": 0.01654,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3.5-flash": {
                "accuracy": 93.026,
                "latency": 24.822,
                "stderr": 0.951,
                "cost_per_test": 0.00137,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 93.026,
                "latency": 28.765,
                "stderr": 0.96,
                "cost_per_test": 0.005674,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "deepseek/deepseek-v4-flash-0731": {
                "accuracy": 92.887,
                "latency": 10.554,
                "stderr": 0.96,
                "cost_per_test": 0.001333,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 92.887,
                "latency": 14.15,
                "stderr": 1.53,
                "cost_per_test": 0.027279,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-4-1-20250805": {
                "accuracy": 92.887,
                "latency": 19.899,
                "stderr": 0.96,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 92.887,
                "latency": 636.88,
                "stderr": 0.96,
                "cost_per_test": 0.012403,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "alibaba/qwen3-max": {
                "accuracy": 92.748,
                "latency": 0.0,
                "stderr": 0.969,
                "cost_per_test": 0.005556,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 92.748,
                "latency": 25.286,
                "stderr": 1.305,
                "cost_per_test": 0.030244,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 92.748,
                "latency": 27.899,
                "stderr": 1.115,
                "cost_per_test": 0.028668,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "minimax/MiniMax-M3": {
                "accuracy": 92.748,
                "latency": 33.196,
                "stderr": 1.01,
                "cost_per_test": 0.002851,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 512000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "openai/o3-2025-04-16": {
                "accuracy": 92.608,
                "latency": 10.958,
                "stderr": 0.977,
                "cost_per_test": 0.008551,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-2.5-pro-exp-03-25": {
                "accuracy": 92.608,
                "latency": 14.363,
                "stderr": 0.977,
                "cost_per_test": 0.0082,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 92.608,
                "latency": 16.66,
                "stderr": 1.018,
                "cost_per_test": 0.016803,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 92.608,
                "latency": 17.242,
                "stderr": 0.977,
                "cost_per_test": 0.01105,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4-0709": {
                "accuracy": 92.608,
                "latency": 41.881,
                "stderr": 0.977,
                "cost_per_test": 0.025477,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-opus-4-1-20250805-thinking": {
                "accuracy": 92.469,
                "latency": 23.454,
                "stderr": 0.986,
                "cost_per_test": 0.104728,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "xiaomi/mimo-v2.5-pro": {
                "accuracy": 92.469,
                "latency": 37.78,
                "stderr": 1.851,
                "cost_per_test": 0.001291,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025-thinking": {
                "accuracy": 92.329,
                "latency": 8.652,
                "stderr": 0.994,
                "cost_per_test": 0.00127,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3-max-preview": {
                "accuracy": 92.329,
                "latency": 35.357,
                "stderr": 0.994,
                "cost_per_test": 0.005165,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "nvidia/nemotron-3-ultra-550b-a55b": {
                "accuracy": 92.19,
                "latency": 14.605,
                "stderr": 1.002,
                "cost_per_test": null,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Nvidia",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025": {
                "accuracy": 92.05,
                "latency": 5.274,
                "stderr": 1.01,
                "cost_per_test": 0.001275,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "zai/glm-4.6": {
                "accuracy": 92.05,
                "latency": 29.233,
                "stderr": 1.01,
                "cost_per_test": 0.004832,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 122880,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "google/gemini-3.5-flash-lite": {
                "accuracy": 91.911,
                "latency": 3.454,
                "stderr": 1.018,
                "cost_per_test": 0.003016,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/o1-2024-12-17": {
                "accuracy": 91.911,
                "latency": 20.622,
                "stderr": 1.018,
                "cost_per_test": 0.093755,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "ant/ling-3.0-flash-2607": {
                "accuracy": 91.771,
                "latency": 8.521,
                "stderr": 1.065,
                "cost_per_test": 0.000309,
                "temperature": 0.6,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "alibaba/qwen3.8-27b": {
                "accuracy": 91.492,
                "latency": 54.143,
                "stderr": 1.042,
                "cost_per_test": 0.00481,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 91.353,
                "latency": 10.41,
                "stderr": 1.05,
                "cost_per_test": 0.003454,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 91.353,
                "latency": 14.45,
                "stderr": 1.08,
                "cost_per_test": 0.002026,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514-thinking": {
                "accuracy": 91.353,
                "latency": 28.672,
                "stderr": 1.05,
                "cost_per_test": 0.030852,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3-max-2026-01-23": {
                "accuracy": 91.353,
                "latency": 88.539,
                "stderr": 1.05,
                "cost_per_test": 0.020461,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514": {
                "accuracy": 91.213,
                "latency": 7.935,
                "stderr": 1.057,
                "cost_per_test": 0.010018,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "fireworks/deepseek-v3p2": {
                "accuracy": 91.213,
                "latency": 19.793,
                "stderr": 1.087,
                "cost_per_test": 0.00025,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "fireworks/deepseek-v3p2-thinking": {
                "accuracy": 91.213,
                "latency": 65.509,
                "stderr": 1.057,
                "cost_per_test": 0.000683,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "grok/grok-3-mini-fast-high-reasoning": {
                "accuracy": 90.934,
                "latency": 9.654,
                "stderr": 1.072,
                "cost_per_test": 0.00126,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "fireworks/deepseek-r1": {
                "accuracy": 90.795,
                "latency": 21.636,
                "stderr": 1.634,
                "cost_per_test": 0.010148,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "fireworks/qwen3-235b-a22b": {
                "accuracy": 90.795,
                "latency": 42.234,
                "stderr": 1.15,
                "cost_per_test": 0.001768,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "grok/grok-4-fast-reasoning": {
                "accuracy": 90.656,
                "latency": 4.835,
                "stderr": 1.087,
                "cost_per_test": 0.001417,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "zai/glm-4.5": {
                "accuracy": 90.656,
                "latency": 80.35,
                "stderr": 1.087,
                "cost_per_test": 0.007079,
                "temperature": 0.6,
                "top_p": 1.0,
                "max_output_tokens": 81920,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025-thinking": {
                "accuracy": 90.516,
                "latency": 7.739,
                "stderr": 1.094,
                "cost_per_test": 0.000294,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "zai/glm-4.7": {
                "accuracy": 90.377,
                "latency": 64.727,
                "stderr": 1.101,
                "cost_per_test": 0.005043,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "openai/o3-mini-2025-01-31": {
                "accuracy": 90.237,
                "latency": 15.13,
                "stderr": 1.108,
                "cost_per_test": 0.009205,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "xiaomi/mimo-v2.5": {
                "accuracy": 90.237,
                "latency": 18.026,
                "stderr": 1.108,
                "cost_per_test": 0.000507,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "anthropic/claude-3-7-sonnet-20250219-thinking": {
                "accuracy": 90.237,
                "latency": 24.131,
                "stderr": 1.108,
                "cost_per_test": 0.030189,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "kimi/kimi-k2-thinking": {
                "accuracy": 90.237,
                "latency": 77.057,
                "stderr": 1.195,
                "cost_per_test": 0.00166,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "fireworks/gpt-oss-120b": {
                "accuracy": 90.098,
                "latency": 18.209,
                "stderr": 1.115,
                "cost_per_test": 0.001015,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "together/moonshotai/Kimi-K2-Instruct": {
                "accuracy": 90.098,
                "latency": 23.45,
                "stderr": 1.115,
                "cost_per_test": 0.003117,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "grok/grok-3-mini-fast-low-reasoning": {
                "accuracy": 89.958,
                "latency": 6.336,
                "stderr": 1.122,
                "cost_per_test": 0.000951,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "low",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-4.1-2025-04-14": {
                "accuracy": 89.958,
                "latency": 9.333,
                "stderr": 1.122,
                "cost_per_test": 0.005755,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-3-5-sonnet-20241022": {
                "accuracy": 89.819,
                "latency": 6.303,
                "stderr": 1.129,
                "cost_per_test": 0.008748,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "fireworks/deepseek-v3-0324": {
                "accuracy": 89.679,
                "latency": 23.473,
                "stderr": 1.143,
                "cost_per_test": 0.001441,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "openai/o4-mini-2025-04-16": {
                "accuracy": 89.4,
                "latency": 8.895,
                "stderr": 1.15,
                "cost_per_test": 0.004693,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 89.261,
                "latency": 30.377,
                "stderr": 1.169,
                "cost_per_test": 0.002294,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "fireworks/llama4-maverick-instruct-basic": {
                "accuracy": 88.982,
                "latency": 4.832,
                "stderr": 1.169,
                "cost_per_test": 0.000689,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "google/gemini-2.0-flash-001": {
                "accuracy": 88.703,
                "latency": 2.946,
                "stderr": 1.182,
                "cost_per_test": 0.000309,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-3-7-sonnet-20250219": {
                "accuracy": 88.703,
                "latency": 6.711,
                "stderr": 1.182,
                "cost_per_test": 0.009595,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "minimax/MiniMax-M2.5": {
                "accuracy": 88.703,
                "latency": 28.855,
                "stderr": 1.189,
                "cost_per_test": 0.003528,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025": {
                "accuracy": 88.563,
                "latency": 2.685,
                "stderr": 1.189,
                "cost_per_test": 0.000451,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "mistralai/mistral-large-2512": {
                "accuracy": 88.563,
                "latency": 12.468,
                "stderr": 1.865,
                "cost_per_test": 0.001387,
                "temperature": 0.065,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 88.563,
                "latency": 18.955,
                "stderr": 1.604,
                "cost_per_test": 0.013073,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-4.1-mini-2025-04-14": {
                "accuracy": 88.285,
                "latency": 3.176,
                "stderr": 1.201,
                "cost_per_test": 0.001175,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-1.5-pro-002": {
                "accuracy": 88.145,
                "latency": 2.688,
                "stderr": 1.207,
                "cost_per_test": 0.0024,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-4o-2024-08-06": {
                "accuracy": 88.145,
                "latency": 7.182,
                "stderr": 1.207,
                "cost_per_test": 0.006568,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-3": {
                "accuracy": 88.145,
                "latency": 14.358,
                "stderr": 1.207,
                "cost_per_test": 0.015289,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "mistralai/mistral-medium-2505": {
                "accuracy": 87.866,
                "latency": 6.356,
                "stderr": 1.305,
                "cost_per_test": 0.001278,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "grok/grok-4-1-fast-non-reasoning": {
                "accuracy": 87.448,
                "latency": 2.646,
                "stderr": 1.237,
                "cost_per_test": 0.000211,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 86.89,
                "latency": 5.551,
                "stderr": 1.601,
                "cost_per_test": 0.000242,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5-nano-2025-08-07": {
                "accuracy": 86.332,
                "latency": 17.004,
                "stderr": 1.283,
                "cost_per_test": 0.001283,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-2-1212": {
                "accuracy": 86.053,
                "latency": 7.621,
                "stderr": 1.294,
                "cost_per_test": 0.005695,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 0,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "together/langston/nim/nvidia/llama-3.3-nemotron-super-49b-v1-42e84561-thinking": {
                "accuracy": 86.053,
                "latency": 36.516,
                "stderr": 1.326,
                "cost_per_test": null,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "fireworks/deepseek-v3": {
                "accuracy": 85.495,
                "latency": 16.683,
                "stderr": 1.315,
                "cost_per_test": 0.001331,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "together/meta-llama/Llama-3.3-70B-Instruct-Turbo": {
                "accuracy": 85.216,
                "latency": 4.749,
                "stderr": 1.326,
                "cost_per_test": 0.001606,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "together/meta-llama/Llama-4-Scout-17B-16E-Instruct": {
                "accuracy": 84.937,
                "latency": 5.538,
                "stderr": 1.336,
                "cost_per_test": 0.000512,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "grok/grok-4-fast-non-reasoning": {
                "accuracy": 84.798,
                "latency": 1.597,
                "stderr": 1.341,
                "cost_per_test": 0.001092,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "mistralai/magistral-small-2509": {
                "accuracy": 84.658,
                "latency": 13.848,
                "stderr": 1.346,
                "cost_per_test": 0.00293,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "poolside/laguna-xs.2": {
                "accuracy": 84.658,
                "latency": 40.598,
                "stderr": 1.818,
                "cost_per_test": 0.000454,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "fireworks/gpt-oss-20b": {
                "accuracy": 84.379,
                "latency": 19.51,
                "stderr": 1.53,
                "cost_per_test": 0.000667,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "mistralai/magistral-medium-2509": {
                "accuracy": 84.24,
                "latency": 18.163,
                "stderr": 1.366,
                "cost_per_test": 0.010747,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "openai/gpt-4o-2024-11-20": {
                "accuracy": 84.1,
                "latency": 8.08,
                "stderr": 1.468,
                "cost_per_test": 0.007273,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "mistralai/mistral-large-2411": {
                "accuracy": 83.821,
                "latency": 5.347,
                "stderr": 1.375,
                "cost_per_test": 0.004702,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "google/gemini-1.5-flash-002": {
                "accuracy": 82.706,
                "latency": 1.504,
                "stderr": 1.421,
                "cost_per_test": 0.000164,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-3-5-haiku-20241022": {
                "accuracy": 81.869,
                "latency": 5.974,
                "stderr": 1.439,
                "cost_per_test": 0.002267,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "cohere/command-a-03-2025": {
                "accuracy": 81.869,
                "latency": 9.774,
                "stderr": 1.643,
                "cost_per_test": 0.007289,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 8000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "mistralai/mistral-small-2503": {
                "accuracy": 81.172,
                "latency": 2.868,
                "stderr": 1.46,
                "cost_per_test": 0.000189,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "together/langston/nim/nvidia/llama-3.3-nemotron-super-49b-v1-42e84561": {
                "accuracy": 80.753,
                "latency": 15.727,
                "stderr": 1.628,
                "cost_per_test": null,
                "temperature": 0.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "openai/gpt-4.1-nano-2025-04-14": {
                "accuracy": 80.474,
                "latency": 1.914,
                "stderr": 1.5,
                "cost_per_test": 0.000292,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-4o-mini-2024-07-18": {
                "accuracy": 80.195,
                "latency": 4.753,
                "stderr": 1.488,
                "cost_per_test": 0.000416,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "poolside/laguna-m.1": {
                "accuracy": 80.056,
                "latency": 97.439,
                "stderr": 1.622,
                "cost_per_test": 0.001191,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "mistralai/mistral-small-2402": {
                "accuracy": 79.637,
                "latency": 3.367,
                "stderr": 1.504,
                "cost_per_test": 0.000439,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "ai21labs/jamba-large-1.6": {
                "accuracy": 71.13,
                "latency": 9.478,
                "stderr": 1.692,
                "cost_per_test": 0.004323,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "mistralai/mistral-medium-3.5": {
                "accuracy": 64.435,
                "latency": 90.757,
                "stderr": 1.788,
                "cost_per_test": 0.03228,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "cohere/command-r-plus": {
                "accuracy": 62.483,
                "latency": 5.069,
                "stderr": 1.808,
                "cost_per_test": 0.00615,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 4000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "ai21labs/jamba-mini-1.6": {
                "accuracy": 25.941,
                "latency": 2.025,
                "stderr": 1.637,
                "cost_per_test": 0.000368,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            }
        },
        "business": {
            "anthropic/claude-fable-5-1": {
                "accuracy": 95.184,
                "latency": 14.985,
                "stderr": 0.762,
                "cost_per_test": 0.070536,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "meta/muse_spark_1_1": {
                "accuracy": 94.55,
                "latency": 20.041,
                "stderr": 0.817,
                "cost_per_test": 0.009893,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "google/gemini-3.8-flash": {
                "accuracy": 93.79,
                "latency": 6.442,
                "stderr": 0.859,
                "cost_per_test": 0.019347,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 93.663,
                "latency": 19.25,
                "stderr": 0.867,
                "cost_per_test": 0.030708,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-5": {
                "accuracy": 93.536,
                "latency": 7.81,
                "stderr": 0.875,
                "cost_per_test": 0.022751,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3.7-max": {
                "accuracy": 93.536,
                "latency": 25.757,
                "stderr": 0.875,
                "cost_per_test": 0.014848,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 93.409,
                "latency": 19.549,
                "stderr": 0.883,
                "cost_per_test": 0.023771,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.6-sol": {
                "accuracy": 93.283,
                "latency": 11.262,
                "stderr": 0.891,
                "cost_per_test": 0.0265,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3.5-flash": {
                "accuracy": 93.283,
                "latency": 17.416,
                "stderr": 0.891,
                "cost_per_test": 0.02368,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-4-7": {
                "accuracy": 93.283,
                "latency": 79.437,
                "stderr": 0.944,
                "cost_per_test": 0.050755,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-4.5": {
                "accuracy": 93.283,
                "latency": 336.555,
                "stderr": 0.891,
                "cost_per_test": 0.012806,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-fable-5": {
                "accuracy": 93.156,
                "latency": 15.86,
                "stderr": 0.899,
                "cost_per_test": 0.054887,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3.8-max": {
                "accuracy": 92.902,
                "latency": 51.654,
                "stderr": 0.914,
                "cost_per_test": 0.016642,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "grok/grok-4.3": {
                "accuracy": 92.902,
                "latency": 417.861,
                "stderr": 0.914,
                "cost_per_test": 0.007727,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-3.7-flash": {
                "accuracy": 92.776,
                "latency": 3.906,
                "stderr": 0.922,
                "cost_per_test": 0.010355,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.6-flash": {
                "accuracy": 92.776,
                "latency": 6.7,
                "stderr": 0.922,
                "cost_per_test": 0.012621,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-4-8": {
                "accuracy": 92.649,
                "latency": 16.722,
                "stderr": 0.929,
                "cost_per_test": 0.041351,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-4.6": {
                "accuracy": 92.649,
                "latency": 33.413,
                "stderr": 0.929,
                "cost_per_test": 0.012798,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 92.269,
                "latency": 31.509,
                "stderr": 0.951,
                "cost_per_test": 0.017383,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 92.269,
                "latency": 45.785,
                "stderr": 0.951,
                "cost_per_test": 0.008359,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "alibaba/qwen3.5-plus-thinking": {
                "accuracy": 92.142,
                "latency": 65.495,
                "stderr": 0.958,
                "cost_per_test": 0.007925,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "kimi/kimi-k3": {
                "accuracy": 92.015,
                "latency": 24.794,
                "stderr": 0.965,
                "cost_per_test": 0.019582,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "meta/muse_spark_1_2": {
                "accuracy": 92.015,
                "latency": 31.992,
                "stderr": 0.972,
                "cost_per_test": 0.00905,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 92.015,
                "latency": 35.966,
                "stderr": 0.965,
                "cost_per_test": 0.021982,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 91.888,
                "latency": 26.047,
                "stderr": 1.219,
                "cost_per_test": 0.028754,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "meta/muse_spark": {
                "accuracy": 91.888,
                "latency": 37.05,
                "stderr": 0.972,
                "cost_per_test": 0.001453,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 91.888,
                "latency": 41.95,
                "stderr": 1.573,
                "cost_per_test": 0.052095,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "zai/glm-5.2": {
                "accuracy": 91.888,
                "latency": 47.4,
                "stderr": 0.972,
                "cost_per_test": 0.012567,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 91.888,
                "latency": 399.135,
                "stderr": 0.972,
                "cost_per_test": 0.0266,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 91.762,
                "latency": 22.267,
                "stderr": 1.441,
                "cost_per_test": 0.045262,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "deepseek/deepseek-v4-flash-0731": {
                "accuracy": 91.635,
                "latency": 15.114,
                "stderr": 0.986,
                "cost_per_test": 0.002306,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 91.635,
                "latency": 34.755,
                "stderr": 1.012,
                "cost_per_test": 0.038321,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-sonnet-5": {
                "accuracy": 91.635,
                "latency": 46.752,
                "stderr": 1.393,
                "cost_per_test": 0.078843,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 91.635,
                "latency": 102.866,
                "stderr": 0.986,
                "cost_per_test": 0.01534,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "thinkingmachines/inkling": {
                "accuracy": 91.508,
                "latency": 73.173,
                "stderr": 0.992,
                "cost_per_test": 0.036474,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "grok/grok-4-0709": {
                "accuracy": 91.508,
                "latency": 124.718,
                "stderr": 0.992,
                "cost_per_test": 0.054675,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 91.381,
                "latency": 6.983,
                "stderr": 0.999,
                "cost_per_test": 0.000925,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 91.381,
                "latency": 8.986,
                "stderr": 0.999,
                "cost_per_test": 0.011664,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "zai/glm-5.3-flash": {
                "accuracy": 91.381,
                "latency": 37.273,
                "stderr": 0.999,
                "cost_per_test": 0.00085,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 91.381,
                "latency": 49.206,
                "stderr": 0.999,
                "cost_per_test": 0.023832,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-4-1-20250805-thinking": {
                "accuracy": 91.128,
                "latency": 22.186,
                "stderr": 1.012,
                "cost_per_test": 0.099964,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "zai/glm-5.3": {
                "accuracy": 91.128,
                "latency": 57.848,
                "stderr": 1.012,
                "cost_per_test": 0.015119,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "zai/glm-5.1": {
                "accuracy": 91.128,
                "latency": 66.948,
                "stderr": 1.012,
                "cost_per_test": 0.012184,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "openai/gpt-5.6-luna": {
                "accuracy": 91.001,
                "latency": 15.056,
                "stderr": 1.086,
                "cost_per_test": 0.002321,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "nvidia/nemotron-3-ultra-550b-a55b": {
                "accuracy": 91.001,
                "latency": 29.998,
                "stderr": 1.019,
                "cost_per_test": null,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Nvidia",
                "harness": null
            },
            "deepseek/deepseek-v4-pro-0813": {
                "accuracy": 91.001,
                "latency": 58.163,
                "stderr": 1.069,
                "cost_per_test": 0.012305,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "alibaba/qwen3-max-2026-01-23": {
                "accuracy": 91.001,
                "latency": 102.176,
                "stderr": 1.019,
                "cost_per_test": 0.024141,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/gpt-5.6-terra": {
                "accuracy": 90.875,
                "latency": 6.029,
                "stderr": 1.025,
                "cost_per_test": 0.006882,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 90.748,
                "latency": 31.453,
                "stderr": 1.032,
                "cost_per_test": 0.006461,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "zai/glm-5-thinking": {
                "accuracy": 90.748,
                "latency": 91.325,
                "stderr": 1.032,
                "cost_per_test": 0.011059,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "google/gemini-3.5-flash-lite": {
                "accuracy": 90.621,
                "latency": 4.543,
                "stderr": 1.038,
                "cost_per_test": 0.004449,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3.8-27b": {
                "accuracy": 90.241,
                "latency": 76.16,
                "stderr": 1.057,
                "cost_per_test": 0.007374,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "minimax/MiniMax-M2.1": {
                "accuracy": 90.114,
                "latency": 38.449,
                "stderr": 1.069,
                "cost_per_test": 0.003023,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "fireworks/deepseek-v3p2-thinking": {
                "accuracy": 90.114,
                "latency": 97.784,
                "stderr": 1.063,
                "cost_per_test": 0.001082,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "alibaba/qwen3-max": {
                "accuracy": 90.047,
                "latency": 0.0,
                "stderr": 1.03,
                "cost_per_test": 0.006394,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/o3-2025-04-16": {
                "accuracy": 89.987,
                "latency": 14.834,
                "stderr": 1.069,
                "cost_per_test": 0.011072,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "xiaomi/mimo-v2.5-pro": {
                "accuracy": 89.987,
                "latency": 72.573,
                "stderr": 1.479,
                "cost_per_test": 0.002321,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025-thinking": {
                "accuracy": 89.861,
                "latency": 8.597,
                "stderr": 1.075,
                "cost_per_test": 0.001312,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 89.861,
                "latency": 19.818,
                "stderr": 1.075,
                "cost_per_test": 0.017077,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 89.861,
                "latency": 42.002,
                "stderr": 1.528,
                "cost_per_test": 0.045443,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "thinkingmachines/inkling-small": {
                "accuracy": 89.861,
                "latency": 77.044,
                "stderr": 1.075,
                "cost_per_test": 0.005233,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 262000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "anthropic/claude-opus-4-1-20250805": {
                "accuracy": 89.734,
                "latency": 19.061,
                "stderr": 1.086,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 89.734,
                "latency": 20.098,
                "stderr": 1.098,
                "cost_per_test": 0.003267,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-4-20250514": {
                "accuracy": 89.354,
                "latency": 8.321,
                "stderr": 1.098,
                "cost_per_test": 0.039973,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3.5-flash": {
                "accuracy": 89.354,
                "latency": 26.597,
                "stderr": 1.098,
                "cost_per_test": 0.001478,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "xiaomi/mimo-v2.5": {
                "accuracy": 89.227,
                "latency": 31.297,
                "stderr": 1.104,
                "cost_per_test": 0.000852,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "google/gemini-2.5-pro-exp-03-25": {
                "accuracy": 89.1,
                "latency": 14.635,
                "stderr": 1.109,
                "cost_per_test": 0.007486,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/o1-2024-12-17": {
                "accuracy": 89.1,
                "latency": 23.018,
                "stderr": 1.109,
                "cost_per_test": 0.100986,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025": {
                "accuracy": 88.973,
                "latency": 5.683,
                "stderr": 1.115,
                "cost_per_test": 0.001306,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "fireworks/deepseek-r1": {
                "accuracy": 88.973,
                "latency": 26.612,
                "stderr": 1.438,
                "cost_per_test": 0.011762,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "ant/ling-3.0-flash-2607": {
                "accuracy": 88.847,
                "latency": 20.02,
                "stderr": 1.126,
                "cost_per_test": 0.000872,
                "temperature": 0.6,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "anthropic/claude-3-7-sonnet-20250219-thinking": {
                "accuracy": 88.847,
                "latency": 27.822,
                "stderr": 1.121,
                "cost_per_test": 0.035756,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 88.847,
                "latency": 43.941,
                "stderr": 1.169,
                "cost_per_test": 0.003205,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 88.72,
                "latency": 24.113,
                "stderr": 1.126,
                "cost_per_test": 0.001053,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 88.593,
                "latency": 23.791,
                "stderr": 1.132,
                "cost_per_test": 0.006643,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3-max-preview": {
                "accuracy": 88.593,
                "latency": 45.611,
                "stderr": 1.132,
                "cost_per_test": 0.005816,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "zai/glm-4.6": {
                "accuracy": 88.466,
                "latency": 49.227,
                "stderr": 1.143,
                "cost_per_test": 0.006347,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 122880,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "minimax/MiniMax-M3": {
                "accuracy": 88.34,
                "latency": 30.784,
                "stderr": 1.164,
                "cost_per_test": 0.003849,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 512000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "zai/glm-4.5": {
                "accuracy": 88.34,
                "latency": 219.493,
                "stderr": 1.143,
                "cost_per_test": 0.010354,
                "temperature": 0.6,
                "top_p": 1.0,
                "max_output_tokens": 81920,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "fireworks/gpt-oss-120b": {
                "accuracy": 87.452,
                "latency": 60.567,
                "stderr": 1.199,
                "cost_per_test": 0.002055,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "anthropic/claude-3-7-sonnet-20250219": {
                "accuracy": 87.199,
                "latency": 5.263,
                "stderr": 1.189,
                "cost_per_test": 0.008106,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "zai/glm-4.7": {
                "accuracy": 87.199,
                "latency": 86.161,
                "stderr": 1.189,
                "cost_per_test": 0.0068,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "openai/o4-mini-2025-04-16": {
                "accuracy": 87.072,
                "latency": 10.359,
                "stderr": 1.194,
                "cost_per_test": 0.005291,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-3-mini-fast-high-reasoning": {
                "accuracy": 87.072,
                "latency": 22.144,
                "stderr": 1.194,
                "cost_per_test": 0.001932,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514-thinking": {
                "accuracy": 86.819,
                "latency": 38.29,
                "stderr": 1.204,
                "cost_per_test": 0.035402,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/o3-mini-2025-01-31": {
                "accuracy": 86.692,
                "latency": 24.632,
                "stderr": 1.209,
                "cost_per_test": 0.012235,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "fireworks/deepseek-v3p2": {
                "accuracy": 86.692,
                "latency": 43.276,
                "stderr": 1.256,
                "cost_per_test": 0.000278,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "fireworks/qwen3-235b-a22b": {
                "accuracy": 86.692,
                "latency": 58.89,
                "stderr": 1.228,
                "cost_per_test": 0.002274,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "minimax/MiniMax-M2.5": {
                "accuracy": 86.565,
                "latency": 45.357,
                "stderr": 1.219,
                "cost_per_test": 0.005087,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "kimi/kimi-k2-thinking": {
                "accuracy": 86.565,
                "latency": 89.584,
                "stderr": 1.349,
                "cost_per_test": 0.001802,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 85.678,
                "latency": 9.108,
                "stderr": 1.445,
                "cost_per_test": 0.000415,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-4.1-2025-04-14": {
                "accuracy": 85.298,
                "latency": 5.622,
                "stderr": 1.261,
                "cost_per_test": 0.00618,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 85.298,
                "latency": 30.544,
                "stderr": 1.364,
                "cost_per_test": 0.179429,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "fireworks/deepseek-v3-0324": {
                "accuracy": 85.044,
                "latency": 18.53,
                "stderr": 1.283,
                "cost_per_test": 0.001292,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "grok/grok-3": {
                "accuracy": 84.791,
                "latency": 8.804,
                "stderr": 1.278,
                "cost_per_test": 0.0131,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "mistralai/mistral-large-2512": {
                "accuracy": 84.664,
                "latency": 11.164,
                "stderr": 1.779,
                "cost_per_test": 0.001253,
                "temperature": 0.065,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "grok/grok-3-mini-fast-low-reasoning": {
                "accuracy": 84.537,
                "latency": 7.615,
                "stderr": 1.287,
                "cost_per_test": 0.000997,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "low",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025-thinking": {
                "accuracy": 84.411,
                "latency": 8.627,
                "stderr": 1.291,
                "cost_per_test": 0.000273,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "together/moonshotai/Kimi-K2-Instruct": {
                "accuracy": 84.284,
                "latency": 13.622,
                "stderr": 1.3,
                "cost_per_test": 0.002074,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025": {
                "accuracy": 83.397,
                "latency": 2.586,
                "stderr": 1.325,
                "cost_per_test": 0.000453,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-3-5-sonnet-20241022": {
                "accuracy": 83.27,
                "latency": 5.569,
                "stderr": 1.329,
                "cost_per_test": 0.007379,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "fireworks/llama4-maverick-instruct-basic": {
                "accuracy": 83.016,
                "latency": 4.606,
                "stderr": 1.337,
                "cost_per_test": 0.000542,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "together/langston/nim/nvidia/llama-3.3-nemotron-super-49b-v1-42e84561": {
                "accuracy": 82.636,
                "latency": 21.458,
                "stderr": 1.397,
                "cost_per_test": null,
                "temperature": 0.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "openai/gpt-5-nano-2025-08-07": {
                "accuracy": 82.51,
                "latency": 22.853,
                "stderr": 1.352,
                "cost_per_test": 0.001792,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-4.1-mini-2025-04-14": {
                "accuracy": 82.129,
                "latency": 3.578,
                "stderr": 1.364,
                "cost_per_test": 0.001132,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 82.003,
                "latency": 19.528,
                "stderr": 1.702,
                "cost_per_test": 0.01488,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-2.0-flash-001": {
                "accuracy": 81.876,
                "latency": 4.674,
                "stderr": 1.371,
                "cost_per_test": 0.000434,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101": {
                "accuracy": 80.862,
                "latency": 6.531,
                "stderr": 1.448,
                "cost_per_test": 0.04245,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "mistralai/mistral-medium-2505": {
                "accuracy": 80.735,
                "latency": 8.464,
                "stderr": 1.435,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "poolside/laguna-xs.2": {
                "accuracy": 80.735,
                "latency": 48.277,
                "stderr": 1.72,
                "cost_per_test": 0.000554,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "google/gemini-1.5-pro-002": {
                "accuracy": 80.355,
                "latency": 3.559,
                "stderr": 1.414,
                "cost_per_test": 0.002419,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "fireworks/gpt-oss-20b": {
                "accuracy": 80.355,
                "latency": 20.238,
                "stderr": 1.538,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "mistralai/magistral-medium-2509": {
                "accuracy": 79.709,
                "latency": 44.91,
                "stderr": 1.213,
                "cost_per_test": 0.01941,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "grok/grok-2-1212": {
                "accuracy": 79.214,
                "latency": 9.121,
                "stderr": 1.445,
                "cost_per_test": 0.005639,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 0,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-4o-2024-08-06": {
                "accuracy": 79.087,
                "latency": 7.893,
                "stderr": 1.448,
                "cost_per_test": 0.006438,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "poolside/laguna-m.1": {
                "accuracy": 79.087,
                "latency": 121.732,
                "stderr": 1.488,
                "cost_per_test": 0.001766,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "grok/grok-4-1-fast-non-reasoning": {
                "accuracy": 78.834,
                "latency": 3.11,
                "stderr": 1.454,
                "cost_per_test": 0.000205,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-4o-2024-11-20": {
                "accuracy": 77.693,
                "latency": 7.518,
                "stderr": 1.533,
                "cost_per_test": 0.006093,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "together/meta-llama/Llama-4-Scout-17B-16E-Instruct": {
                "accuracy": 75.665,
                "latency": 4.955,
                "stderr": 1.528,
                "cost_per_test": 0.000424,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "together/langston/nim/nvidia/llama-3.3-nemotron-super-49b-v1-42e84561-thinking": {
                "accuracy": 75.665,
                "latency": 29.982,
                "stderr": 1.546,
                "cost_per_test": null,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "mistralai/magistral-small-2509": {
                "accuracy": 75.539,
                "latency": 11.181,
                "stderr": 1.53,
                "cost_per_test": 0.003029,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "fireworks/deepseek-v3": {
                "accuracy": 74.271,
                "latency": 4.416,
                "stderr": 1.556,
                "cost_per_test": 0.001053,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "together/meta-llama/Llama-3.3-70B-Instruct-Turbo": {
                "accuracy": 73.384,
                "latency": 3.293,
                "stderr": 1.573,
                "cost_per_test": 0.001252,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "cohere/command-a-03-2025": {
                "accuracy": 73.384,
                "latency": 7.866,
                "stderr": 1.678,
                "cost_per_test": 0.006893,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 8000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "grok/grok-4-fast-non-reasoning": {
                "accuracy": 72.877,
                "latency": 2.094,
                "stderr": 1.583,
                "cost_per_test": 0.000895,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-4.1-nano-2025-04-14": {
                "accuracy": 72.37,
                "latency": 1.992,
                "stderr": 1.634,
                "cost_per_test": 0.000272,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "mistralai/mistral-large-2411": {
                "accuracy": 71.736,
                "latency": 7.912,
                "stderr": 1.603,
                "cost_per_test": 0.004557,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "google/gemini-1.5-flash-002": {
                "accuracy": 71.61,
                "latency": 1.793,
                "stderr": 1.607,
                "cost_per_test": 0.000151,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "mistralai/mistral-small-2402": {
                "accuracy": 70.596,
                "latency": 5.154,
                "stderr": 1.622,
                "cost_per_test": 0.000479,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "openai/gpt-4o-mini-2024-07-18": {
                "accuracy": 70.342,
                "latency": 5.379,
                "stderr": 1.626,
                "cost_per_test": 0.000405,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-3-5-haiku-20241022": {
                "accuracy": 69.835,
                "latency": 5.48,
                "stderr": 1.634,
                "cost_per_test": 0.001917,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "mistralai/mistral-small-2503": {
                "accuracy": 69.328,
                "latency": 4.059,
                "stderr": 1.642,
                "cost_per_test": 0.000198,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "grok/grok-4-fast-reasoning": {
                "accuracy": 69.328,
                "latency": 7.248,
                "stderr": 1.642,
                "cost_per_test": 0.001573,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "mistralai/mistral-medium-3.5": {
                "accuracy": 64.385,
                "latency": 101.826,
                "stderr": 1.705,
                "cost_per_test": 0.047867,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "ai21labs/jamba-large-1.6": {
                "accuracy": 47.275,
                "latency": 9.096,
                "stderr": 1.774,
                "cost_per_test": 0.004523,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "cohere/command-r-plus": {
                "accuracy": 41.572,
                "latency": 5.995,
                "stderr": 1.754,
                "cost_per_test": 0.00596,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 4000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514": {
                "accuracy": 34.728,
                "latency": 10.899,
                "stderr": 1.695,
                "cost_per_test": 0.008309,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "ai21labs/jamba-mini-1.6": {
                "accuracy": 30.418,
                "latency": 2.722,
                "stderr": 1.634,
                "cost_per_test": 0.000353,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            }
        },
        "chemistry": {
            "anthropic/claude-fable-5": {
                "accuracy": 93.905,
                "latency": 39.622,
                "stderr": 0.711,
                "cost_per_test": 0.094911,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-fable-5-1": {
                "accuracy": 93.816,
                "latency": 18.464,
                "stderr": 0.716,
                "cost_per_test": 0.095732,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "meta/muse_spark_1_1": {
                "accuracy": 93.375,
                "latency": 22.142,
                "stderr": 0.744,
                "cost_per_test": 0.011437,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "anthropic/claude-opus-5": {
                "accuracy": 93.286,
                "latency": 9.971,
                "stderr": 0.744,
                "cost_per_test": 0.031874,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 93.286,
                "latency": 36.067,
                "stderr": 0.744,
                "cost_per_test": 0.044854,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 93.286,
                "latency": 67.055,
                "stderr": 1.074,
                "cost_per_test": 0.077784,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-4-7": {
                "accuracy": 93.286,
                "latency": 549.138,
                "stderr": 0.744,
                "cost_per_test": 0.104175,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.5-flash": {
                "accuracy": 93.11,
                "latency": 55.325,
                "stderr": 0.77,
                "cost_per_test": 0.035628,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-4-8": {
                "accuracy": 92.845,
                "latency": 29.316,
                "stderr": 0.766,
                "cost_per_test": 0.081956,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 92.845,
                "latency": 33.183,
                "stderr": 0.766,
                "cost_per_test": 0.053294,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "meta/muse_spark_1_2": {
                "accuracy": 92.845,
                "latency": 45.052,
                "stderr": 0.766,
                "cost_per_test": 0.011447,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "google/gemini-3.8-flash": {
                "accuracy": 92.491,
                "latency": 8.465,
                "stderr": 0.783,
                "cost_per_test": 0.027138,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3.8-max": {
                "accuracy": 92.491,
                "latency": 64.407,
                "stderr": 0.783,
                "cost_per_test": 0.019826,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 92.491,
                "latency": 211.305,
                "stderr": 0.808,
                "cost_per_test": 0.02316,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 92.491,
                "latency": 588.835,
                "stderr": 0.787,
                "cost_per_test": 0.041657,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "google/gemini-3.7-flash": {
                "accuracy": 92.314,
                "latency": 5.24,
                "stderr": 0.792,
                "cost_per_test": 0.014862,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.6-flash": {
                "accuracy": 92.314,
                "latency": 9.267,
                "stderr": 0.792,
                "cost_per_test": 0.018427,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3.7-max": {
                "accuracy": 92.314,
                "latency": 40.784,
                "stderr": 0.792,
                "cost_per_test": 0.024737,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "alibaba/qwen3.5-plus-thinking": {
                "accuracy": 92.226,
                "latency": 92.131,
                "stderr": 0.796,
                "cost_per_test": 0.011862,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "grok/grok-4.6": {
                "accuracy": 92.049,
                "latency": 51.733,
                "stderr": 0.804,
                "cost_per_test": 0.017551,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-5.6-sol": {
                "accuracy": 91.961,
                "latency": 14.047,
                "stderr": 0.808,
                "cost_per_test": 0.038842,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 91.873,
                "latency": 50.721,
                "stderr": 0.812,
                "cost_per_test": 0.02887,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 91.873,
                "latency": 73.25,
                "stderr": 1.369,
                "cost_per_test": 0.099424,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "zai/glm-5.1": {
                "accuracy": 91.844,
                "latency": 126.749,
                "stderr": 0.815,
                "cost_per_test": 0.02417,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "kimi/kimi-k3": {
                "accuracy": 91.784,
                "latency": 57.728,
                "stderr": 0.816,
                "cost_per_test": 0.029196,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "alibaba/qwen3.8-27b": {
                "accuracy": 91.784,
                "latency": 115.333,
                "stderr": 0.816,
                "cost_per_test": 0.011081,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "nvidia/nemotron-3-ultra-550b-a55b": {
                "accuracy": 91.696,
                "latency": 48.434,
                "stderr": 0.82,
                "cost_per_test": null,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Nvidia",
                "harness": null
            },
            "deepseek/deepseek-v4-pro-0813": {
                "accuracy": 91.696,
                "latency": 71.74,
                "stderr": 0.824,
                "cost_per_test": 0.015709,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "zai/glm-5.2": {
                "accuracy": 91.608,
                "latency": 75.947,
                "stderr": 0.824,
                "cost_per_test": 0.020916,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "grok/grok-4.5": {
                "accuracy": 91.608,
                "latency": 273.466,
                "stderr": 0.824,
                "cost_per_test": 0.018195,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "grok/grok-4.3": {
                "accuracy": 91.608,
                "latency": 307.571,
                "stderr": 0.824,
                "cost_per_test": 0.012042,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "meta/muse_spark": {
                "accuracy": 91.519,
                "latency": 50.435,
                "stderr": 0.828,
                "cost_per_test": 0.001905,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 91.431,
                "latency": 66.874,
                "stderr": 0.832,
                "cost_per_test": 0.013391,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "minimax/MiniMax-M2.1": {
                "accuracy": 91.343,
                "latency": 54.018,
                "stderr": 0.836,
                "cost_per_test": 0.004516,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "zai/glm-5.3": {
                "accuracy": 91.343,
                "latency": 61.175,
                "stderr": 0.836,
                "cost_per_test": 0.01539,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "thinkingmachines/inkling": {
                "accuracy": 91.343,
                "latency": 86.144,
                "stderr": 0.836,
                "cost_per_test": 0.046349,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 91.254,
                "latency": 54.56,
                "stderr": 0.84,
                "cost_per_test": 0.028489,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 91.254,
                "latency": 56.981,
                "stderr": 0.847,
                "cost_per_test": 0.011818,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 91.078,
                "latency": 21.94,
                "stderr": 1.082,
                "cost_per_test": 0.048378,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-5.3-flash": {
                "accuracy": 91.078,
                "latency": 30.116,
                "stderr": 0.847,
                "cost_per_test": 0.000699,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "alibaba/qwen3.5-flash": {
                "accuracy": 91.078,
                "latency": 39.771,
                "stderr": 0.847,
                "cost_per_test": 0.002165,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 91.078,
                "latency": 45.357,
                "stderr": 1.151,
                "cost_per_test": 0.055856,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.5-flash-lite": {
                "accuracy": 90.989,
                "latency": 6.086,
                "stderr": 0.851,
                "cost_per_test": 0.006593,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 90.989,
                "latency": 12.463,
                "stderr": 0.851,
                "cost_per_test": 0.00158,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.6-luna": {
                "accuracy": 90.989,
                "latency": 15.973,
                "stderr": 0.935,
                "cost_per_test": 0.002791,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "deepseek/deepseek-v4-flash-0731": {
                "accuracy": 90.989,
                "latency": 18.154,
                "stderr": 0.851,
                "cost_per_test": 0.00332,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "alibaba/qwen3-max-2026-01-23": {
                "accuracy": 90.989,
                "latency": 133.733,
                "stderr": 0.851,
                "cost_per_test": 0.034771,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "anthropic/claude-opus-4-1-20250805-thinking": {
                "accuracy": 90.869,
                "latency": 38.648,
                "stderr": 0.868,
                "cost_per_test": 0.18329,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-sonnet-5": {
                "accuracy": 90.813,
                "latency": 9.099,
                "stderr": 1.062,
                "cost_per_test": 0.022172,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 90.813,
                "latency": 47.605,
                "stderr": 1.205,
                "cost_per_test": 0.304988,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "zai/glm-5-thinking": {
                "accuracy": 90.813,
                "latency": 115.706,
                "stderr": 0.859,
                "cost_per_test": 0.017438,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 90.724,
                "latency": 11.56,
                "stderr": 0.862,
                "cost_per_test": 0.016085,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 90.724,
                "latency": 35.206,
                "stderr": 0.873,
                "cost_per_test": 0.041726,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "thinkingmachines/inkling-small": {
                "accuracy": 90.724,
                "latency": 100.705,
                "stderr": 0.862,
                "cost_per_test": 0.006994,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 262000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "google/gemini-2.5-pro-exp-03-25": {
                "accuracy": 90.548,
                "latency": 22.87,
                "stderr": 0.87,
                "cost_per_test": 0.011033,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "minimax/MiniMax-M3": {
                "accuracy": 90.548,
                "latency": 50.811,
                "stderr": 0.912,
                "cost_per_test": 0.007038,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 512000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "ant/ling-3.0-flash-2607": {
                "accuracy": 90.459,
                "latency": 26.979,
                "stderr": 0.908,
                "cost_per_test": 0.001114,
                "temperature": 0.6,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "xiaomi/mimo-v2.5-pro": {
                "accuracy": 90.459,
                "latency": 108.606,
                "stderr": 1.038,
                "cost_per_test": 0.003601,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "grok/grok-4-0709": {
                "accuracy": 90.459,
                "latency": 135.856,
                "stderr": 0.873,
                "cost_per_test": 0.070869,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "alibaba/qwen3-max": {
                "accuracy": 90.283,
                "latency": 0.0,
                "stderr": 0.88,
                "cost_per_test": 0.008748,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/gpt-5.6-terra": {
                "accuracy": 90.194,
                "latency": 7.923,
                "stderr": 0.884,
                "cost_per_test": 0.009964,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 90.106,
                "latency": 28.924,
                "stderr": 0.887,
                "cost_per_test": 0.024697,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3-max-preview": {
                "accuracy": 89.929,
                "latency": 48.823,
                "stderr": 0.894,
                "cost_per_test": 0.008645,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025-thinking": {
                "accuracy": 89.841,
                "latency": 13.498,
                "stderr": 0.898,
                "cost_per_test": 0.002146,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "xiaomi/mimo-v2.5": {
                "accuracy": 89.841,
                "latency": 42.513,
                "stderr": 0.898,
                "cost_per_test": 0.001299,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 89.841,
                "latency": 43.47,
                "stderr": 0.898,
                "cost_per_test": 0.032013,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "fireworks/deepseek-v3p2-thinking": {
                "accuracy": 89.841,
                "latency": 127.304,
                "stderr": 0.898,
                "cost_per_test": 0.001651,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514-thinking": {
                "accuracy": 89.664,
                "latency": 56.436,
                "stderr": 0.908,
                "cost_per_test": 0.061301,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101": {
                "accuracy": 89.488,
                "latency": 8.301,
                "stderr": 1.108,
                "cost_per_test": 0.064481,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-4-1-20250805": {
                "accuracy": 89.488,
                "latency": 18.038,
                "stderr": 0.912,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "zai/glm-4.7": {
                "accuracy": 89.223,
                "latency": 160.189,
                "stderr": 0.922,
                "cost_per_test": 0.013335,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 89.134,
                "latency": 15.401,
                "stderr": 0.928,
                "cost_per_test": 0.003744,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/o3-2025-04-16": {
                "accuracy": 89.134,
                "latency": 22.617,
                "stderr": 0.925,
                "cost_per_test": 0.016523,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-3-mini-fast-high-reasoning": {
                "accuracy": 89.134,
                "latency": 27.757,
                "stderr": 0.925,
                "cost_per_test": 0.002691,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "zai/glm-4.6": {
                "accuracy": 89.134,
                "latency": 53.391,
                "stderr": 0.928,
                "cost_per_test": 0.007878,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 122880,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025": {
                "accuracy": 89.046,
                "latency": 9.256,
                "stderr": 0.928,
                "cost_per_test": 0.002146,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 89.046,
                "latency": 21.705,
                "stderr": 0.928,
                "cost_per_test": 0.0064,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 88.869,
                "latency": 66.867,
                "stderr": 0.938,
                "cost_per_test": 0.005376,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "anthropic/claude-3-7-sonnet-20250219-thinking": {
                "accuracy": 88.516,
                "latency": 44.105,
                "stderr": 0.948,
                "cost_per_test": 0.060384,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "fireworks/deepseek-r1": {
                "accuracy": 88.428,
                "latency": 38.388,
                "stderr": 1.38,
                "cost_per_test": 0.017823,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "fireworks/qwen3-235b-a22b": {
                "accuracy": 88.163,
                "latency": 59.35,
                "stderr": 0.978,
                "cost_per_test": 0.003506,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "anthropic/claude-opus-4-20250514": {
                "accuracy": 87.898,
                "latency": 11.11,
                "stderr": 0.969,
                "cost_per_test": 0.062945,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "fireworks/gpt-oss-120b": {
                "accuracy": 87.809,
                "latency": 27.523,
                "stderr": 0.978,
                "cost_per_test": 0.001869,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "minimax/MiniMax-M2.5": {
                "accuracy": 87.721,
                "latency": 60.449,
                "stderr": 0.978,
                "cost_per_test": 0.006569,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 87.633,
                "latency": 37.247,
                "stderr": 0.978,
                "cost_per_test": 0.001572,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "grok/grok-3-mini-fast-low-reasoning": {
                "accuracy": 87.456,
                "latency": 14.197,
                "stderr": 0.984,
                "cost_per_test": 0.001312,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "low",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "fireworks/deepseek-v3p2": {
                "accuracy": 87.191,
                "latency": 42.326,
                "stderr": 1.005,
                "cost_per_test": 0.000418,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025-thinking": {
                "accuracy": 87.102,
                "latency": 11.67,
                "stderr": 0.996,
                "cost_per_test": 0.000436,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 87.014,
                "latency": 9.297,
                "stderr": 1.122,
                "cost_per_test": 0.00054,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-3": {
                "accuracy": 86.749,
                "latency": 18.732,
                "stderr": 1.008,
                "cost_per_test": 0.020273,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/o3-mini-2025-01-31": {
                "accuracy": 86.749,
                "latency": 24.487,
                "stderr": 1.008,
                "cost_per_test": 0.015137,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/o4-mini-2025-04-16": {
                "accuracy": 86.661,
                "latency": 10.034,
                "stderr": 1.011,
                "cost_per_test": 0.006322,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/o1-2024-12-17": {
                "accuracy": 86.661,
                "latency": 34.973,
                "stderr": 1.013,
                "cost_per_test": 0.152426,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "kimi/kimi-k2-thinking": {
                "accuracy": 86.307,
                "latency": 102.14,
                "stderr": 1.162,
                "cost_per_test": 0.002417,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "zai/glm-4.5": {
                "accuracy": 86.131,
                "latency": 350.263,
                "stderr": 1.027,
                "cost_per_test": 0.016233,
                "temperature": 0.6,
                "top_p": 1.0,
                "max_output_tokens": 81920,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "fireworks/llama4-maverick-instruct-basic": {
                "accuracy": 85.866,
                "latency": 6.509,
                "stderr": 1.035,
                "cost_per_test": 0.00085,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514": {
                "accuracy": 85.777,
                "latency": 12.501,
                "stderr": 1.038,
                "cost_per_test": 0.013022,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5-nano-2025-08-07": {
                "accuracy": 85.424,
                "latency": 27.441,
                "stderr": 1.049,
                "cost_per_test": 0.00228,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025": {
                "accuracy": 85.336,
                "latency": 4.106,
                "stderr": 1.051,
                "cost_per_test": 0.000798,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 85.247,
                "latency": 31.291,
                "stderr": 1.482,
                "cost_per_test": 0.026179,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "mistralai/mistral-large-2512": {
                "accuracy": 84.982,
                "latency": 15.68,
                "stderr": 1.375,
                "cost_per_test": 0.001859,
                "temperature": 0.065,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "openai/gpt-4.1-mini-2025-04-14": {
                "accuracy": 83.657,
                "latency": 5.939,
                "stderr": 1.099,
                "cost_per_test": 0.001817,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-2.0-flash-001": {
                "accuracy": 83.127,
                "latency": 6.964,
                "stderr": 1.113,
                "cost_per_test": 0.000652,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-4.1-2025-04-14": {
                "accuracy": 82.951,
                "latency": 20.885,
                "stderr": 1.118,
                "cost_per_test": 0.010238,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "fireworks/deepseek-v3-0324": {
                "accuracy": 82.951,
                "latency": 47.977,
                "stderr": 1.14,
                "cost_per_test": 0.001779,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "anthropic/claude-3-7-sonnet-20250219": {
                "accuracy": 82.686,
                "latency": 6.869,
                "stderr": 1.125,
                "cost_per_test": 0.012016,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "together/moonshotai/Kimi-K2-Instruct": {
                "accuracy": 82.067,
                "latency": 27.216,
                "stderr": 1.14,
                "cost_per_test": 0.003645,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "fireworks/gpt-oss-20b": {
                "accuracy": 80.389,
                "latency": 31.536,
                "stderr": 1.306,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "anthropic/claude-3-5-sonnet-20241022": {
                "accuracy": 79.77,
                "latency": 6.262,
                "stderr": 1.194,
                "cost_per_test": 0.010358,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "mistralai/mistral-medium-2505": {
                "accuracy": 79.594,
                "latency": 15.685,
                "stderr": 1.204,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "google/gemini-1.5-pro-002": {
                "accuracy": 78.975,
                "latency": 5.361,
                "stderr": 1.211,
                "cost_per_test": 0.003754,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-2-1212": {
                "accuracy": 78.71,
                "latency": 16.668,
                "stderr": 1.217,
                "cost_per_test": 0.008818,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 0,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "grok/grok-4-1-fast-non-reasoning": {
                "accuracy": 77.297,
                "latency": 4.232,
                "stderr": 1.245,
                "cost_per_test": 0.000291,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "grok/grok-4-fast-reasoning": {
                "accuracy": 76.855,
                "latency": 10.218,
                "stderr": 1.254,
                "cost_per_test": 0.002341,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "fireworks/deepseek-v3": {
                "accuracy": 75.972,
                "latency": 17.212,
                "stderr": 1.27,
                "cost_per_test": 0.001588,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "together/meta-llama/Llama-4-Scout-17B-16E-Instruct": {
                "accuracy": 75.353,
                "latency": 6.093,
                "stderr": 1.281,
                "cost_per_test": 0.000616,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "openai/gpt-4o-2024-11-20": {
                "accuracy": 75.265,
                "latency": 13.454,
                "stderr": 1.315,
                "cost_per_test": 0.010192,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-4o-2024-08-06": {
                "accuracy": 73.322,
                "latency": 14.513,
                "stderr": 1.315,
                "cost_per_test": 0.009883,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "mistralai/mistral-medium-3.5": {
                "accuracy": 73.057,
                "latency": 97.072,
                "stderr": 1.319,
                "cost_per_test": 0.0594,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "google/gemini-1.5-flash-002": {
                "accuracy": 71.555,
                "latency": 2.534,
                "stderr": 1.341,
                "cost_per_test": 0.00023,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "mistralai/mistral-large-2411": {
                "accuracy": 71.113,
                "latency": 12.738,
                "stderr": 1.347,
                "cost_per_test": 0.007141,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "together/langston/nim/nvidia/llama-3.3-nemotron-super-49b-v1-42e84561-thinking": {
                "accuracy": 70.848,
                "latency": 52.813,
                "stderr": 1.367,
                "cost_per_test": null,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "grok/grok-4-fast-non-reasoning": {
                "accuracy": 70.406,
                "latency": 1.581,
                "stderr": 1.357,
                "cost_per_test": 0.001271,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "together/meta-llama/Llama-3.3-70B-Instruct-Turbo": {
                "accuracy": 70.406,
                "latency": 4.962,
                "stderr": 1.357,
                "cost_per_test": 0.001937,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "poolside/laguna-xs.2": {
                "accuracy": 69.611,
                "latency": 87.951,
                "stderr": 1.463,
                "cost_per_test": 0.001183,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "cohere/command-a-03-2025": {
                "accuracy": 68.905,
                "latency": 17.157,
                "stderr": 1.404,
                "cost_per_test": 0.010878,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 8000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "mistralai/mistral-small-2402": {
                "accuracy": 68.11,
                "latency": 8.57,
                "stderr": 1.385,
                "cost_per_test": 0.000727,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "poolside/laguna-m.1": {
                "accuracy": 67.226,
                "latency": 168.987,
                "stderr": 1.417,
                "cost_per_test": 0.003468,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "mistralai/mistral-small-2503": {
                "accuracy": 66.608,
                "latency": 5.882,
                "stderr": 1.402,
                "cost_per_test": 0.000295,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "mistralai/magistral-small-2509": {
                "accuracy": 66.608,
                "latency": 22.217,
                "stderr": 1.402,
                "cost_per_test": 0.005379,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "together/langston/nim/nvidia/llama-3.3-nemotron-super-49b-v1-42e84561": {
                "accuracy": 65.724,
                "latency": 19.731,
                "stderr": 1.423,
                "cost_per_test": null,
                "temperature": 0.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "openai/gpt-4.1-nano-2025-04-14": {
                "accuracy": 64.223,
                "latency": 3.327,
                "stderr": 1.44,
                "cost_per_test": 0.000433,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-4o-mini-2024-07-18": {
                "accuracy": 62.279,
                "latency": 8.068,
                "stderr": 1.441,
                "cost_per_test": 0.000573,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-3-5-haiku-20241022": {
                "accuracy": 61.219,
                "latency": 5.68,
                "stderr": 1.448,
                "cost_per_test": 0.002622,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "ai21labs/jamba-large-1.6": {
                "accuracy": 50.265,
                "latency": 18.791,
                "stderr": 1.486,
                "cost_per_test": 0.007081,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "mistralai/magistral-medium-2509": {
                "accuracy": 40.283,
                "latency": 36.238,
                "stderr": 1.458,
                "cost_per_test": 0.018854,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "cohere/command-r-plus": {
                "accuracy": 28.71,
                "latency": 6.703,
                "stderr": 1.345,
                "cost_per_test": 0.007669,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 4000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "ai21labs/jamba-mini-1.6": {
                "accuracy": 22.085,
                "latency": 4.486,
                "stderr": 1.231,
                "cost_per_test": 0.000561,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            }
        },
        "computer_science": {
            "anthropic/claude-fable-5-1": {
                "accuracy": 94.878,
                "latency": 14.297,
                "stderr": 1.089,
                "cost_per_test": 0.072746,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-fable-5": {
                "accuracy": 94.146,
                "latency": 17.469,
                "stderr": 1.159,
                "cost_per_test": 0.08492,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-5": {
                "accuracy": 93.902,
                "latency": 8.895,
                "stderr": 1.182,
                "cost_per_test": 0.028057,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.8-flash": {
                "accuracy": 92.683,
                "latency": 8.75,
                "stderr": 1.286,
                "cost_per_test": 0.022638,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-4-8": {
                "accuracy": 92.683,
                "latency": 19.219,
                "stderr": 1.286,
                "cost_per_test": 0.056235,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 92.683,
                "latency": 21.033,
                "stderr": 1.286,
                "cost_per_test": 0.025749,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-4-7": {
                "accuracy": 92.683,
                "latency": 97.967,
                "stderr": 1.286,
                "cost_per_test": 0.042102,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.7-flash": {
                "accuracy": 91.951,
                "latency": 4.823,
                "stderr": 1.344,
                "cost_per_test": 0.010676,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.6-flash": {
                "accuracy": 91.707,
                "latency": 7.006,
                "stderr": 1.362,
                "cost_per_test": 0.013111,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.5-flash": {
                "accuracy": 91.707,
                "latency": 10.896,
                "stderr": 1.362,
                "cost_per_test": 0.022972,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-4.6": {
                "accuracy": 91.707,
                "latency": 39.856,
                "stderr": 1.362,
                "cost_per_test": 0.014017,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 91.463,
                "latency": 22.896,
                "stderr": 1.38,
                "cost_per_test": 0.035823,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "kimi/kimi-k3": {
                "accuracy": 91.22,
                "latency": 25.445,
                "stderr": 1.398,
                "cost_per_test": 0.020224,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "alibaba/qwen3.7-max": {
                "accuracy": 91.22,
                "latency": 28.331,
                "stderr": 1.398,
                "cost_per_test": 0.016489,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 91.22,
                "latency": 154.119,
                "stderr": 1.398,
                "cost_per_test": 0.017371,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "openai/gpt-5.6-terra": {
                "accuracy": 90.976,
                "latency": 6.133,
                "stderr": 1.415,
                "cost_per_test": 0.008055,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 90.976,
                "latency": 36.425,
                "stderr": 1.415,
                "cost_per_test": 0.019599,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3.5-plus-thinking": {
                "accuracy": 90.976,
                "latency": 62.583,
                "stderr": 1.415,
                "cost_per_test": 0.008357,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "alibaba/qwen3.8-max": {
                "accuracy": 90.976,
                "latency": 65.158,
                "stderr": 1.415,
                "cost_per_test": 0.017793,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "grok/grok-4.5": {
                "accuracy": 90.976,
                "latency": 495.996,
                "stderr": 1.415,
                "cost_per_test": 0.013831,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "meta/muse_spark_1_1": {
                "accuracy": 90.732,
                "latency": 18.999,
                "stderr": 1.498,
                "cost_per_test": 0.009764,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "openai/gpt-5.6-sol": {
                "accuracy": 90.732,
                "latency": 20.3,
                "stderr": 1.432,
                "cost_per_test": 0.030318,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "meta/muse_spark_1_2": {
                "accuracy": 90.732,
                "latency": 36.959,
                "stderr": 1.432,
                "cost_per_test": 0.009355,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 90.732,
                "latency": 48.764,
                "stderr": 2.303,
                "cost_per_test": 0.052137,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "deepseek/deepseek-v4-flash-0731": {
                "accuracy": 90.244,
                "latency": 16.508,
                "stderr": 1.465,
                "cost_per_test": 0.002655,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "zai/glm-5.2": {
                "accuracy": 90.244,
                "latency": 43.156,
                "stderr": 1.465,
                "cost_per_test": 0.012207,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "deepseek/deepseek-v4-pro-0813": {
                "accuracy": 90.0,
                "latency": 46.177,
                "stderr": 1.588,
                "cost_per_test": 0.011516,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 90.0,
                "latency": 49.213,
                "stderr": 2.382,
                "cost_per_test": 0.060432,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "meta/muse_spark": {
                "accuracy": 90.0,
                "latency": 51.459,
                "stderr": 1.482,
                "cost_per_test": 0.001545,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "openai/gpt-5.6-luna": {
                "accuracy": 89.756,
                "latency": 13.524,
                "stderr": 1.513,
                "cost_per_test": 0.002273,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 89.756,
                "latency": 21.483,
                "stderr": 1.498,
                "cost_per_test": 0.018843,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 89.756,
                "latency": 22.64,
                "stderr": 2.274,
                "cost_per_test": 0.045736,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "nvidia/nemotron-3-ultra-550b-a55b": {
                "accuracy": 89.756,
                "latency": 31.773,
                "stderr": 1.498,
                "cost_per_test": null,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Nvidia",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 89.756,
                "latency": 38.434,
                "stderr": 1.498,
                "cost_per_test": 0.027059,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 89.756,
                "latency": 44.734,
                "stderr": 1.498,
                "cost_per_test": 0.008378,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "anthropic/claude-opus-4-1-20250805": {
                "accuracy": 89.512,
                "latency": 14.003,
                "stderr": 1.573,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 89.512,
                "latency": 40.257,
                "stderr": 2.142,
                "cost_per_test": 0.043611,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "zai/glm-5.1": {
                "accuracy": 89.487,
                "latency": 44.147,
                "stderr": 1.517,
                "cost_per_test": 0.007687,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "anthropic/claude-sonnet-5": {
                "accuracy": 89.024,
                "latency": 19.153,
                "stderr": 1.559,
                "cost_per_test": 0.016148,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-4-1-20250805-thinking": {
                "accuracy": 89.024,
                "latency": 27.701,
                "stderr": 1.643,
                "cost_per_test": 0.129006,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 89.024,
                "latency": 34.835,
                "stderr": 1.544,
                "cost_per_test": 0.001091,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 89.024,
                "latency": 39.138,
                "stderr": 1.544,
                "cost_per_test": 0.007604,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "zai/glm-5.3": {
                "accuracy": 89.024,
                "latency": 43.81,
                "stderr": 1.544,
                "cost_per_test": 0.012424,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "grok/grok-4.3": {
                "accuracy": 89.024,
                "latency": 505.088,
                "stderr": 1.544,
                "cost_per_test": 0.00878,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 89.024,
                "latency": 753.864,
                "stderr": 1.559,
                "cost_per_test": 0.029423,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "zai/glm-5.3-flash": {
                "accuracy": 88.78,
                "latency": 41.108,
                "stderr": 1.559,
                "cost_per_test": 0.000949,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "thinkingmachines/inkling-small": {
                "accuracy": 88.78,
                "latency": 60.904,
                "stderr": 1.559,
                "cost_per_test": 0.004781,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 262000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "thinkingmachines/inkling": {
                "accuracy": 88.537,
                "latency": 60.965,
                "stderr": 1.573,
                "cost_per_test": 0.030874,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "alibaba/qwen3-max-2026-01-23": {
                "accuracy": 88.537,
                "latency": 104.941,
                "stderr": 1.573,
                "cost_per_test": 0.024617,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "alibaba/qwen3-max": {
                "accuracy": 88.293,
                "latency": 0.0,
                "stderr": 1.588,
                "cost_per_test": 0.006896,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "anthropic/claude-opus-4-20250514": {
                "accuracy": 88.293,
                "latency": 10.875,
                "stderr": 1.588,
                "cost_per_test": 0.054997,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 88.293,
                "latency": 33.394,
                "stderr": 1.588,
                "cost_per_test": 0.034376,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "minimax/MiniMax-M3": {
                "accuracy": 88.293,
                "latency": 38.531,
                "stderr": 1.683,
                "cost_per_test": 0.004813,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 512000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "alibaba/qwen3.8-27b": {
                "accuracy": 88.293,
                "latency": 174.301,
                "stderr": 1.588,
                "cost_per_test": 0.009725,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "xiaomi/mimo-v2.5-pro": {
                "accuracy": 88.049,
                "latency": 58.443,
                "stderr": 2.268,
                "cost_per_test": 0.001837,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 87.805,
                "latency": 30.035,
                "stderr": 1.616,
                "cost_per_test": 0.021717,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3.5-flash-lite": {
                "accuracy": 87.561,
                "latency": 4.763,
                "stderr": 1.63,
                "cost_per_test": 0.004473,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 87.561,
                "latency": 10.573,
                "stderr": 1.63,
                "cost_per_test": 0.012189,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 87.561,
                "latency": 21.213,
                "stderr": 1.63,
                "cost_per_test": 0.000995,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "alibaba/qwen3.5-flash": {
                "accuracy": 87.561,
                "latency": 32.653,
                "stderr": 1.63,
                "cost_per_test": 0.001708,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "zai/glm-5-thinking": {
                "accuracy": 87.561,
                "latency": 75.826,
                "stderr": 1.63,
                "cost_per_test": 0.010569,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025": {
                "accuracy": 87.317,
                "latency": 7.075,
                "stderr": 1.643,
                "cost_per_test": 0.001549,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "ant/ling-3.0-flash-2607": {
                "accuracy": 87.317,
                "latency": 18.325,
                "stderr": 1.643,
                "cost_per_test": 0.000683,
                "temperature": 0.6,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "fireworks/deepseek-v3p2-thinking": {
                "accuracy": 87.317,
                "latency": 87.233,
                "stderr": 1.643,
                "cost_per_test": 0.001014,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025-thinking": {
                "accuracy": 87.073,
                "latency": 10.423,
                "stderr": 1.657,
                "cost_per_test": 0.001574,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/o3-2025-04-16": {
                "accuracy": 87.073,
                "latency": 15.171,
                "stderr": 1.657,
                "cost_per_test": 0.011691,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 87.073,
                "latency": 24.636,
                "stderr": 1.683,
                "cost_per_test": 0.003661,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "minimax/MiniMax-M2.1": {
                "accuracy": 87.073,
                "latency": 38.487,
                "stderr": 1.67,
                "cost_per_test": 0.003263,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "alibaba/qwen3-max-preview": {
                "accuracy": 87.073,
                "latency": 44.122,
                "stderr": 1.657,
                "cost_per_test": 0.006141,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514": {
                "accuracy": 86.829,
                "latency": 8.295,
                "stderr": 1.67,
                "cost_per_test": 0.011196,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 86.829,
                "latency": 15.009,
                "stderr": 1.67,
                "cost_per_test": 0.004107,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "xiaomi/mimo-v2.5": {
                "accuracy": 86.585,
                "latency": 45.123,
                "stderr": 1.683,
                "cost_per_test": 0.000767,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "grok/grok-4-0709": {
                "accuracy": 86.585,
                "latency": 46.115,
                "stderr": 1.683,
                "cost_per_test": 0.050302,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "zai/glm-4.7": {
                "accuracy": 86.585,
                "latency": 103.442,
                "stderr": 1.696,
                "cost_per_test": 0.007504,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "zai/glm-4.6": {
                "accuracy": 86.341,
                "latency": 50.687,
                "stderr": 1.709,
                "cost_per_test": 0.006772,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 122880,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 86.098,
                "latency": 43.448,
                "stderr": 1.709,
                "cost_per_test": 0.00347,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "zai/glm-4.5": {
                "accuracy": 86.098,
                "latency": 162.109,
                "stderr": 1.709,
                "cost_per_test": 0.009198,
                "temperature": 0.6,
                "top_p": 1.0,
                "max_output_tokens": 81920,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "google/gemini-2.5-pro-exp-03-25": {
                "accuracy": 85.854,
                "latency": 19.727,
                "stderr": 1.721,
                "cost_per_test": 0.009532,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/o1-2024-12-17": {
                "accuracy": 85.854,
                "latency": 28.946,
                "stderr": 1.721,
                "cost_per_test": 0.119714,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "minimax/MiniMax-M2.5": {
                "accuracy": 85.854,
                "latency": 40.754,
                "stderr": 1.721,
                "cost_per_test": 0.005007,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 85.61,
                "latency": 41.97,
                "stderr": 2.114,
                "cost_per_test": 0.237579,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "fireworks/deepseek-v3p2": {
                "accuracy": 85.61,
                "latency": 62.738,
                "stderr": 1.826,
                "cost_per_test": 0.000259,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "fireworks/deepseek-r1": {
                "accuracy": 85.366,
                "latency": 26.57,
                "stderr": 2.298,
                "cost_per_test": 0.012518,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "fireworks/qwen3-235b-a22b": {
                "accuracy": 85.366,
                "latency": 45.551,
                "stderr": 1.781,
                "cost_per_test": 0.002233,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "openai/gpt-4.1-2025-04-14": {
                "accuracy": 85.122,
                "latency": 5.688,
                "stderr": 1.758,
                "cost_per_test": 0.00729,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "fireworks/gpt-oss-120b": {
                "accuracy": 85.122,
                "latency": 20.726,
                "stderr": 1.758,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514-thinking": {
                "accuracy": 85.122,
                "latency": 55.944,
                "stderr": 1.758,
                "cost_per_test": 0.052996,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/o4-mini-2025-04-16": {
                "accuracy": 84.878,
                "latency": 11.322,
                "stderr": 1.769,
                "cost_per_test": 0.005717,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-3-7-sonnet-20250219-thinking": {
                "accuracy": 84.878,
                "latency": 34.525,
                "stderr": 1.769,
                "cost_per_test": 0.044185,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 84.634,
                "latency": 6.365,
                "stderr": 1.899,
                "cost_per_test": 0.000369,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/o3-mini-2025-01-31": {
                "accuracy": 84.634,
                "latency": 18.417,
                "stderr": 1.781,
                "cost_per_test": 0.01223,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101": {
                "accuracy": 84.146,
                "latency": 8.62,
                "stderr": 1.975,
                "cost_per_test": 0.057543,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-3": {
                "accuracy": 84.146,
                "latency": 10.499,
                "stderr": 1.804,
                "cost_per_test": 0.016725,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "grok/grok-3-mini-fast-high-reasoning": {
                "accuracy": 84.146,
                "latency": 15.194,
                "stderr": 1.804,
                "cost_per_test": 0.001848,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "grok/grok-3-mini-fast-low-reasoning": {
                "accuracy": 83.902,
                "latency": 7.254,
                "stderr": 1.815,
                "cost_per_test": 0.001104,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "low",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "kimi/kimi-k2-thinking": {
                "accuracy": 83.902,
                "latency": 112.665,
                "stderr": 1.919,
                "cost_per_test": 0.002135,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "anthropic/claude-3-7-sonnet-20250219": {
                "accuracy": 83.659,
                "latency": 6.094,
                "stderr": 1.826,
                "cost_per_test": 0.010214,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5-nano-2025-08-07": {
                "accuracy": 83.415,
                "latency": 17.478,
                "stderr": 1.837,
                "cost_per_test": 0.001542,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "mistralai/mistral-large-2512": {
                "accuracy": 83.171,
                "latency": 11.872,
                "stderr": 2.247,
                "cost_per_test": 0.00154,
                "temperature": 0.065,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "together/moonshotai/Kimi-K2-Instruct": {
                "accuracy": 83.171,
                "latency": 27.239,
                "stderr": 1.848,
                "cost_per_test": 0.003545,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "openai/gpt-4.1-mini-2025-04-14": {
                "accuracy": 82.927,
                "latency": 4.076,
                "stderr": 1.858,
                "cost_per_test": 0.001393,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "mistralai/mistral-medium-3.5": {
                "accuracy": 82.927,
                "latency": 50.137,
                "stderr": 1.858,
                "cost_per_test": 0.046075,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "fireworks/deepseek-v3-0324": {
                "accuracy": 82.195,
                "latency": 29.083,
                "stderr": 1.948,
                "cost_per_test": 0.001713,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025": {
                "accuracy": 81.951,
                "latency": 3.595,
                "stderr": 1.899,
                "cost_per_test": 0.00063,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025-thinking": {
                "accuracy": 81.951,
                "latency": 9.315,
                "stderr": 1.899,
                "cost_per_test": 0.000345,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 81.707,
                "latency": 28.835,
                "stderr": 2.398,
                "cost_per_test": 0.020126,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "fireworks/llama4-maverick-instruct-basic": {
                "accuracy": 81.463,
                "latency": 4.686,
                "stderr": 1.919,
                "cost_per_test": 0.000735,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "anthropic/claude-3-5-sonnet-20241022": {
                "accuracy": 81.22,
                "latency": 6.402,
                "stderr": 1.929,
                "cost_per_test": 0.009551,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "fireworks/gpt-oss-20b": {
                "accuracy": 81.22,
                "latency": 23.628,
                "stderr": 2.036,
                "cost_per_test": 0.001057,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "google/gemini-2.0-flash-001": {
                "accuracy": 80.488,
                "latency": 6.226,
                "stderr": 1.957,
                "cost_per_test": 0.000593,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-4-fast-reasoning": {
                "accuracy": 80.244,
                "latency": 7.272,
                "stderr": 1.966,
                "cost_per_test": 0.001847,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "grok/grok-4-1-fast-non-reasoning": {
                "accuracy": 78.537,
                "latency": 4.302,
                "stderr": 2.052,
                "cost_per_test": 0.000237,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "mistralai/mistral-medium-2505": {
                "accuracy": 78.293,
                "latency": 8.798,
                "stderr": 2.044,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "google/gemini-1.5-pro-002": {
                "accuracy": 78.049,
                "latency": 3.692,
                "stderr": 2.044,
                "cost_per_test": 0.003045,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-4o-2024-08-06": {
                "accuracy": 78.049,
                "latency": 10.061,
                "stderr": 2.044,
                "cost_per_test": 0.007916,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "fireworks/deepseek-v3": {
                "accuracy": 77.317,
                "latency": 10.47,
                "stderr": 2.068,
                "cost_per_test": 0.001581,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "together/langston/nim/nvidia/llama-3.3-nemotron-super-49b-v1-42e84561": {
                "accuracy": 76.341,
                "latency": 16.809,
                "stderr": 2.162,
                "cost_per_test": null,
                "temperature": 0.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "grok/grok-2-1212": {
                "accuracy": 76.098,
                "latency": 10.101,
                "stderr": 2.106,
                "cost_per_test": 0.006592,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 0,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-4o-2024-11-20": {
                "accuracy": 75.854,
                "latency": 9.572,
                "stderr": 2.258,
                "cost_per_test": 0.008429,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "mistralai/mistral-large-2411": {
                "accuracy": 75.61,
                "latency": 7.355,
                "stderr": 2.121,
                "cost_per_test": 0.005473,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "mistralai/magistral-small-2509": {
                "accuracy": 74.39,
                "latency": 13.435,
                "stderr": 2.156,
                "cost_per_test": 0.003485,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "together/langston/nim/nvidia/llama-3.3-nemotron-super-49b-v1-42e84561-thinking": {
                "accuracy": 74.146,
                "latency": 34.963,
                "stderr": 2.162,
                "cost_per_test": null,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "together/meta-llama/Llama-3.3-70B-Instruct-Turbo": {
                "accuracy": 72.927,
                "latency": 4.266,
                "stderr": 2.194,
                "cost_per_test": 0.001692,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "poolside/laguna-xs.2": {
                "accuracy": 72.927,
                "latency": 44.341,
                "stderr": 2.342,
                "cost_per_test": 0.000456,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "poolside/laguna-m.1": {
                "accuracy": 72.683,
                "latency": 62.839,
                "stderr": 2.253,
                "cost_per_test": 0.001775,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "mistralai/mistral-small-2503": {
                "accuracy": 71.951,
                "latency": 3.282,
                "stderr": 2.219,
                "cost_per_test": 0.000219,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "cohere/command-a-03-2025": {
                "accuracy": 70.976,
                "latency": 8.315,
                "stderr": 2.368,
                "cost_per_test": 0.008203,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 8000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "mistralai/magistral-medium-2509": {
                "accuracy": 70.0,
                "latency": 27.638,
                "stderr": 2.263,
                "cost_per_test": 0.014657,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "together/meta-llama/Llama-4-Scout-17B-16E-Instruct": {
                "accuracy": 69.756,
                "latency": 4.796,
                "stderr": 2.268,
                "cost_per_test": 0.000542,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "grok/grok-4-fast-non-reasoning": {
                "accuracy": 67.64,
                "latency": 1.945,
                "stderr": 2.308,
                "cost_per_test": 0.001211,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "mistralai/mistral-small-2402": {
                "accuracy": 67.561,
                "latency": 4.094,
                "stderr": 2.312,
                "cost_per_test": 0.000531,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "google/gemini-1.5-flash-002": {
                "accuracy": 66.098,
                "latency": 1.769,
                "stderr": 2.346,
                "cost_per_test": 0.000191,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-3-5-haiku-20241022": {
                "accuracy": 65.61,
                "latency": 5.583,
                "stderr": 2.346,
                "cost_per_test": 0.002407,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-4o-mini-2024-07-18": {
                "accuracy": 64.39,
                "latency": 5.662,
                "stderr": 2.365,
                "cost_per_test": 0.000462,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-4.1-nano-2025-04-14": {
                "accuracy": 63.659,
                "latency": 2.425,
                "stderr": 2.385,
                "cost_per_test": 0.000351,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "ai21labs/jamba-large-1.6": {
                "accuracy": 52.683,
                "latency": 6.983,
                "stderr": 2.466,
                "cost_per_test": 0.004989,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "cohere/command-r-plus": {
                "accuracy": 46.585,
                "latency": 4.718,
                "stderr": 2.463,
                "cost_per_test": 0.006329,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 4000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "ai21labs/jamba-mini-1.6": {
                "accuracy": 28.293,
                "latency": 2.478,
                "stderr": 2.224,
                "cost_per_test": 0.00044,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            }
        },
        "economics": {
            "anthropic/claude-fable-5-1": {
                "accuracy": 95.142,
                "latency": 11.834,
                "stderr": 0.74,
                "cost_per_test": 0.057274,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-fable-5": {
                "accuracy": 93.839,
                "latency": 12.804,
                "stderr": 0.828,
                "cost_per_test": 0.052862,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-5": {
                "accuracy": 93.72,
                "latency": 7.939,
                "stderr": 0.835,
                "cost_per_test": 0.022734,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 93.365,
                "latency": 14.452,
                "stderr": 0.857,
                "cost_per_test": 0.017048,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-4-7": {
                "accuracy": 92.299,
                "latency": 609.105,
                "stderr": 0.955,
                "cost_per_test": 0.036308,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.8-flash": {
                "accuracy": 92.062,
                "latency": 5.685,
                "stderr": 0.931,
                "cost_per_test": 0.016353,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.5-flash": {
                "accuracy": 91.943,
                "latency": 8.784,
                "stderr": 0.937,
                "cost_per_test": 0.017242,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 91.943,
                "latency": 17.613,
                "stderr": 0.937,
                "cost_per_test": 0.026377,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-4-8": {
                "accuracy": 91.825,
                "latency": 17.061,
                "stderr": 0.943,
                "cost_per_test": 0.04071,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-4.6": {
                "accuracy": 91.825,
                "latency": 68.761,
                "stderr": 0.943,
                "cost_per_test": 0.01053,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-3.7-flash": {
                "accuracy": 91.706,
                "latency": 3.142,
                "stderr": 0.949,
                "cost_per_test": 0.008178,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3.7-max": {
                "accuracy": 91.706,
                "latency": 20.211,
                "stderr": 0.949,
                "cost_per_test": 0.011905,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "kimi/kimi-k3": {
                "accuracy": 91.588,
                "latency": 22.695,
                "stderr": 0.955,
                "cost_per_test": 0.011738,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "grok/grok-4.5": {
                "accuracy": 91.351,
                "latency": 325.874,
                "stderr": 0.968,
                "cost_per_test": 0.009717,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "meta/muse_spark_1_1": {
                "accuracy": 91.232,
                "latency": 15.915,
                "stderr": 1.014,
                "cost_per_test": 0.008745,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "meta/muse_spark_1_2": {
                "accuracy": 91.232,
                "latency": 29.458,
                "stderr": 0.985,
                "cost_per_test": 0.006835,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 90.995,
                "latency": 25.017,
                "stderr": 0.985,
                "cost_per_test": 0.012976,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.6-sol": {
                "accuracy": 90.64,
                "latency": 6.99,
                "stderr": 1.003,
                "cost_per_test": 0.018513,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 90.64,
                "latency": 24.868,
                "stderr": 1.528,
                "cost_per_test": 0.029023,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.6-flash": {
                "accuracy": 90.521,
                "latency": 5.287,
                "stderr": 1.008,
                "cost_per_test": 0.009457,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "zai/glm-5.2": {
                "accuracy": 90.403,
                "latency": 27.0,
                "stderr": 1.014,
                "cost_per_test": 0.007413,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "zai/glm-5.3-flash": {
                "accuracy": 90.284,
                "latency": 21.06,
                "stderr": 1.019,
                "cost_per_test": 0.000508,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 90.284,
                "latency": 21.494,
                "stderr": 1.684,
                "cost_per_test": 0.022582,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "minimax/MiniMax-M2.1": {
                "accuracy": 90.284,
                "latency": 21.514,
                "stderr": 1.019,
                "cost_per_test": 0.001944,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 90.166,
                "latency": 25.255,
                "stderr": 1.305,
                "cost_per_test": 0.024978,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-4.3": {
                "accuracy": 90.166,
                "latency": 413.591,
                "stderr": 1.025,
                "cost_per_test": 0.005946,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-opus-4-1-20250805-thinking": {
                "accuracy": 90.071,
                "latency": 19.787,
                "stderr": 1.028,
                "cost_per_test": 0.08772,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 89.929,
                "latency": 28.846,
                "stderr": 1.036,
                "cost_per_test": 0.01703,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3.8-max": {
                "accuracy": 89.929,
                "latency": 31.834,
                "stderr": 1.036,
                "cost_per_test": 0.008894,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 89.929,
                "latency": 82.769,
                "stderr": 1.036,
                "cost_per_test": 0.01056,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "google/gemini-3.5-flash-lite": {
                "accuracy": 89.81,
                "latency": 3.282,
                "stderr": 1.041,
                "cost_per_test": 0.00297,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "meta/muse_spark": {
                "accuracy": 89.81,
                "latency": 34.11,
                "stderr": 1.041,
                "cost_per_test": 0.001362,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "thinkingmachines/inkling": {
                "accuracy": 89.81,
                "latency": 40.923,
                "stderr": 1.041,
                "cost_per_test": 0.020592,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 89.692,
                "latency": 6.504,
                "stderr": 1.047,
                "cost_per_test": 0.008769,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-opus-4-20250514": {
                "accuracy": 89.692,
                "latency": 9.348,
                "stderr": 1.047,
                "cost_per_test": 0.043404,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 89.692,
                "latency": 15.82,
                "stderr": 1.356,
                "cost_per_test": 0.030999,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 89.692,
                "latency": 35.519,
                "stderr": 1.047,
                "cost_per_test": 0.006698,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "thinkingmachines/inkling-small": {
                "accuracy": 89.692,
                "latency": 36.032,
                "stderr": 1.047,
                "cost_per_test": 0.002849,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 262000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "alibaba/qwen3.5-plus-thinking": {
                "accuracy": 89.692,
                "latency": 65.942,
                "stderr": 1.047,
                "cost_per_test": 0.006729,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101": {
                "accuracy": 89.573,
                "latency": 7.509,
                "stderr": 1.068,
                "cost_per_test": 0.044699,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 89.573,
                "latency": 18.946,
                "stderr": 1.052,
                "cost_per_test": 0.014362,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 89.573,
                "latency": 20.197,
                "stderr": 1.062,
                "cost_per_test": 0.020471,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-4-1-20250805": {
                "accuracy": 89.455,
                "latency": 18.5,
                "stderr": 1.057,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-sonnet-5": {
                "accuracy": 89.455,
                "latency": 27.727,
                "stderr": 1.189,
                "cost_per_test": 0.045893,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 89.336,
                "latency": 6.013,
                "stderr": 1.062,
                "cost_per_test": 0.000798,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "deepseek/deepseek-v4-flash-0731": {
                "accuracy": 89.336,
                "latency": 8.395,
                "stderr": 1.062,
                "cost_per_test": 0.001361,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "alibaba/qwen3.5-flash": {
                "accuracy": 89.336,
                "latency": 24.247,
                "stderr": 1.062,
                "cost_per_test": 0.001311,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "deepseek/deepseek-v4-pro-0813": {
                "accuracy": 89.218,
                "latency": 38.278,
                "stderr": 1.088,
                "cost_per_test": 0.007669,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "openai/gpt-5.6-terra": {
                "accuracy": 88.981,
                "latency": 4.591,
                "stderr": 1.078,
                "cost_per_test": 0.005342,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/o3-2025-04-16": {
                "accuracy": 88.981,
                "latency": 9.432,
                "stderr": 1.078,
                "cost_per_test": 0.007898,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 88.981,
                "latency": 18.018,
                "stderr": 1.078,
                "cost_per_test": 0.010654,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 88.981,
                "latency": 24.478,
                "stderr": 1.312,
                "cost_per_test": 0.130533,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "zai/glm-5.1": {
                "accuracy": 88.981,
                "latency": 28.009,
                "stderr": 1.078,
                "cost_per_test": 0.00488,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "zai/glm-5.3": {
                "accuracy": 88.981,
                "latency": 39.042,
                "stderr": 1.078,
                "cost_per_test": 0.009673,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 88.863,
                "latency": 24.636,
                "stderr": 1.083,
                "cost_per_test": 0.004975,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 88.863,
                "latency": 373.31,
                "stderr": 1.083,
                "cost_per_test": 0.014327,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 88.744,
                "latency": 15.064,
                "stderr": 1.093,
                "cost_per_test": 0.00222,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4-0709": {
                "accuracy": 88.744,
                "latency": 49.714,
                "stderr": 1.088,
                "cost_per_test": 0.028301,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "nvidia/nemotron-3-ultra-550b-a55b": {
                "accuracy": 88.507,
                "latency": 16.623,
                "stderr": 1.098,
                "cost_per_test": null,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Nvidia",
                "harness": null
            },
            "fireworks/deepseek-v3p2-thinking": {
                "accuracy": 88.507,
                "latency": 62.365,
                "stderr": 1.098,
                "cost_per_test": 0.000679,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514": {
                "accuracy": 88.152,
                "latency": 10.177,
                "stderr": 1.112,
                "cost_per_test": 0.008757,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3.8-27b": {
                "accuracy": 88.152,
                "latency": 15.205,
                "stderr": 1.112,
                "cost_per_test": 0.005109,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/gpt-5.6-luna": {
                "accuracy": 88.033,
                "latency": 11.143,
                "stderr": 1.141,
                "cost_per_test": 0.001524,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "minimax/MiniMax-M3": {
                "accuracy": 87.915,
                "latency": 24.193,
                "stderr": 1.136,
                "cost_per_test": 0.002542,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 512000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "xiaomi/mimo-v2.5-pro": {
                "accuracy": 87.915,
                "latency": 42.503,
                "stderr": 1.675,
                "cost_per_test": 0.001218,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "zai/glm-5-thinking": {
                "accuracy": 87.915,
                "latency": 71.636,
                "stderr": 1.122,
                "cost_per_test": 0.007021,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "google/gemini-2.5-pro-exp-03-25": {
                "accuracy": 87.796,
                "latency": 13.595,
                "stderr": 1.127,
                "cost_per_test": 0.007895,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3-max-2026-01-23": {
                "accuracy": 87.796,
                "latency": 83.049,
                "stderr": 1.127,
                "cost_per_test": 0.019045,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 87.678,
                "latency": 13.023,
                "stderr": 1.131,
                "cost_per_test": 0.000626,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "fireworks/deepseek-r1": {
                "accuracy": 87.678,
                "latency": 21.108,
                "stderr": 1.474,
                "cost_per_test": 0.009453,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514-thinking": {
                "accuracy": 87.678,
                "latency": 32.561,
                "stderr": 1.131,
                "cost_per_test": 0.027652,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025": {
                "accuracy": 87.441,
                "latency": 5.184,
                "stderr": 1.141,
                "cost_per_test": 0.001202,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025-thinking": {
                "accuracy": 87.441,
                "latency": 8.176,
                "stderr": 1.141,
                "cost_per_test": 0.001203,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3-max": {
                "accuracy": 87.322,
                "latency": 0.0,
                "stderr": 1.145,
                "cost_per_test": 0.00436,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "alibaba/qwen3-max-preview": {
                "accuracy": 87.322,
                "latency": 32.749,
                "stderr": 1.145,
                "cost_per_test": 0.004664,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/o1-2024-12-17": {
                "accuracy": 87.204,
                "latency": 20.018,
                "stderr": 1.15,
                "cost_per_test": 0.088583,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-4.5": {
                "accuracy": 86.967,
                "latency": 76.691,
                "stderr": 1.159,
                "cost_per_test": 0.006371,
                "temperature": 0.6,
                "top_p": 1.0,
                "max_output_tokens": 81920,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "zai/glm-4.7": {
                "accuracy": 86.848,
                "latency": 64.194,
                "stderr": 1.163,
                "cost_per_test": 0.00463,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "ant/ling-3.0-flash-2607": {
                "accuracy": 86.73,
                "latency": 11.729,
                "stderr": 1.168,
                "cost_per_test": 0.000414,
                "temperature": 0.6,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "xiaomi/mimo-v2.5": {
                "accuracy": 86.73,
                "latency": 18.403,
                "stderr": 1.177,
                "cost_per_test": 0.000494,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "fireworks/deepseek-v3p2": {
                "accuracy": 86.73,
                "latency": 19.105,
                "stderr": 1.288,
                "cost_per_test": 0.00021,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 86.73,
                "latency": 27.505,
                "stderr": 1.168,
                "cost_per_test": 0.00379,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-3-7-sonnet-20250219-thinking": {
                "accuracy": 86.493,
                "latency": 21.682,
                "stderr": 1.177,
                "cost_per_test": 0.026329,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "zai/glm-4.6": {
                "accuracy": 86.493,
                "latency": 36.579,
                "stderr": 1.177,
                "cost_per_test": 0.005003,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 122880,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "fireworks/qwen3-235b-a22b": {
                "accuracy": 86.256,
                "latency": 47.446,
                "stderr": 1.223,
                "cost_per_test": 0.001695,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "fireworks/llama4-maverick-instruct-basic": {
                "accuracy": 86.019,
                "latency": 5.183,
                "stderr": 1.194,
                "cost_per_test": 0.000626,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "openai/gpt-4.1-2025-04-14": {
                "accuracy": 86.019,
                "latency": 5.216,
                "stderr": 1.194,
                "cost_per_test": 0.004358,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-3-mini-fast-high-reasoning": {
                "accuracy": 86.019,
                "latency": 8.779,
                "stderr": 1.194,
                "cost_per_test": 0.001172,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-3-7-sonnet-20250219": {
                "accuracy": 85.9,
                "latency": 5.639,
                "stderr": 1.198,
                "cost_per_test": 0.008379,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "mistralai/mistral-large-2512": {
                "accuracy": 85.782,
                "latency": 11.545,
                "stderr": 1.662,
                "cost_per_test": 0.001255,
                "temperature": 0.065,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "grok/grok-3-mini-fast-low-reasoning": {
                "accuracy": 85.308,
                "latency": 5.538,
                "stderr": 1.219,
                "cost_per_test": 0.000863,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "low",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/o4-mini-2025-04-16": {
                "accuracy": 85.308,
                "latency": 6.938,
                "stderr": 1.219,
                "cost_per_test": 0.004229,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "together/moonshotai/Kimi-K2-Instruct": {
                "accuracy": 85.308,
                "latency": 15.862,
                "stderr": 1.219,
                "cost_per_test": 0.002448,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 85.308,
                "latency": 16.328,
                "stderr": 1.715,
                "cost_per_test": 0.010968,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "minimax/MiniMax-M2.5": {
                "accuracy": 85.308,
                "latency": 29.01,
                "stderr": 1.223,
                "cost_per_test": 0.003273,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "openai/o3-mini-2025-01-31": {
                "accuracy": 85.071,
                "latency": 15.538,
                "stderr": 1.227,
                "cost_per_test": 0.009259,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "kimi/kimi-k2-thinking": {
                "accuracy": 85.071,
                "latency": 71.495,
                "stderr": 1.329,
                "cost_per_test": 0.001543,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 84.953,
                "latency": 26.258,
                "stderr": 1.25,
                "cost_per_test": 0.002168,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "grok/grok-3": {
                "accuracy": 84.834,
                "latency": 8.42,
                "stderr": 1.235,
                "cost_per_test": 0.013966,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025-thinking": {
                "accuracy": 84.716,
                "latency": 8.223,
                "stderr": 1.239,
                "cost_per_test": 0.000285,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-4.1-mini-2025-04-14": {
                "accuracy": 84.597,
                "latency": 2.854,
                "stderr": 1.243,
                "cost_per_test": 0.001048,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025": {
                "accuracy": 84.479,
                "latency": 2.493,
                "stderr": 1.246,
                "cost_per_test": 0.000428,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "fireworks/deepseek-v3-0324": {
                "accuracy": 84.479,
                "latency": 12.961,
                "stderr": 1.246,
                "cost_per_test": 0.001274,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "grok/grok-4-fast-reasoning": {
                "accuracy": 84.242,
                "latency": 4.536,
                "stderr": 1.254,
                "cost_per_test": 0.001322,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "fireworks/gpt-oss-120b": {
                "accuracy": 84.242,
                "latency": 42.971,
                "stderr": 1.266,
                "cost_per_test": 0.001145,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "anthropic/claude-3-5-sonnet-20241022": {
                "accuracy": 84.123,
                "latency": 5.825,
                "stderr": 1.258,
                "cost_per_test": 0.007887,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-2.0-flash-001": {
                "accuracy": 83.649,
                "latency": 2.017,
                "stderr": 1.273,
                "cost_per_test": 0.000231,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-4-1-fast-non-reasoning": {
                "accuracy": 82.583,
                "latency": 1.975,
                "stderr": 1.305,
                "cost_per_test": 0.000171,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "poolside/laguna-xs.2": {
                "accuracy": 81.872,
                "latency": 34.574,
                "stderr": 1.677,
                "cost_per_test": 0.000387,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "openai/gpt-5-nano-2025-08-07": {
                "accuracy": 81.517,
                "latency": 17.711,
                "stderr": 1.336,
                "cost_per_test": 0.001325,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "mistralai/magistral-medium-2509": {
                "accuracy": 81.398,
                "latency": 43.071,
                "stderr": 1.343,
                "cost_per_test": 0.012351,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "openai/gpt-4o-2024-11-20": {
                "accuracy": 81.28,
                "latency": 7.483,
                "stderr": 1.407,
                "cost_per_test": 0.00664,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 81.161,
                "latency": 6.062,
                "stderr": 1.413,
                "cost_per_test": 0.000222,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "mistralai/mistral-medium-2505": {
                "accuracy": 81.161,
                "latency": 6.655,
                "stderr": 1.374,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "grok/grok-2-1212": {
                "accuracy": 80.924,
                "latency": 5.024,
                "stderr": 1.352,
                "cost_per_test": 0.004702,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 0,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "together/langston/nim/nvidia/llama-3.3-nemotron-super-49b-v1-42e84561-thinking": {
                "accuracy": 80.924,
                "latency": 34.704,
                "stderr": 1.371,
                "cost_per_test": null,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "fireworks/deepseek-v3": {
                "accuracy": 80.806,
                "latency": 4.392,
                "stderr": 1.356,
                "cost_per_test": 0.001227,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "google/gemini-1.5-pro-002": {
                "accuracy": 80.687,
                "latency": 2.351,
                "stderr": 1.359,
                "cost_per_test": 0.002155,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "poolside/laguna-m.1": {
                "accuracy": 80.687,
                "latency": 88.62,
                "stderr": 1.407,
                "cost_per_test": 0.001115,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "together/meta-llama/Llama-3.3-70B-Instruct-Turbo": {
                "accuracy": 79.858,
                "latency": 3.751,
                "stderr": 1.381,
                "cost_per_test": 0.001456,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "openai/gpt-4o-2024-08-06": {
                "accuracy": 79.858,
                "latency": 6.954,
                "stderr": 1.381,
                "cost_per_test": 0.006085,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "together/langston/nim/nvidia/llama-3.3-nemotron-super-49b-v1-42e84561": {
                "accuracy": 79.858,
                "latency": 15.644,
                "stderr": 1.531,
                "cost_per_test": null,
                "temperature": 0.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "together/meta-llama/Llama-4-Scout-17B-16E-Instruct": {
                "accuracy": 78.91,
                "latency": 4.802,
                "stderr": 1.404,
                "cost_per_test": 0.000468,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "fireworks/gpt-oss-20b": {
                "accuracy": 78.436,
                "latency": 14.323,
                "stderr": 1.535,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "grok/grok-4-fast-non-reasoning": {
                "accuracy": 78.318,
                "latency": 2.074,
                "stderr": 1.418,
                "cost_per_test": 0.001013,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "mistralai/mistral-medium-3.5": {
                "accuracy": 78.199,
                "latency": 60.82,
                "stderr": 1.421,
                "cost_per_test": 0.032339,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "cohere/command-a-03-2025": {
                "accuracy": 77.725,
                "latency": 5.885,
                "stderr": 1.535,
                "cost_per_test": 0.006694,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 8000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "mistralai/mistral-large-2411": {
                "accuracy": 76.066,
                "latency": 5.326,
                "stderr": 1.469,
                "cost_per_test": 0.004166,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "mistralai/magistral-small-2509": {
                "accuracy": 76.066,
                "latency": 11.336,
                "stderr": 1.469,
                "cost_per_test": 0.00282,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "google/gemini-1.5-flash-002": {
                "accuracy": 73.934,
                "latency": 1.376,
                "stderr": 1.513,
                "cost_per_test": 0.000149,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-3-5-haiku-20241022": {
                "accuracy": 73.697,
                "latency": 5.701,
                "stderr": 1.516,
                "cost_per_test": 0.002098,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "mistralai/mistral-small-2503": {
                "accuracy": 73.578,
                "latency": 2.73,
                "stderr": 1.518,
                "cost_per_test": 0.000167,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "openai/gpt-4o-mini-2024-07-18": {
                "accuracy": 73.341,
                "latency": 4.476,
                "stderr": 1.522,
                "cost_per_test": 0.000352,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-4.1-nano-2025-04-14": {
                "accuracy": 73.223,
                "latency": 2.07,
                "stderr": 1.562,
                "cost_per_test": 0.00028,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "mistralai/mistral-small-2402": {
                "accuracy": 71.327,
                "latency": 3.519,
                "stderr": 1.557,
                "cost_per_test": 0.000429,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "ai21labs/jamba-large-1.6": {
                "accuracy": 55.45,
                "latency": 5.624,
                "stderr": 1.713,
                "cost_per_test": 0.004016,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "cohere/command-r-plus": {
                "accuracy": 52.251,
                "latency": 4.52,
                "stderr": 1.72,
                "cost_per_test": 0.005444,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 4000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "ai21labs/jamba-mini-1.6": {
                "accuracy": 29.621,
                "latency": 1.406,
                "stderr": 1.57,
                "cost_per_test": 0.00032,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            }
        },
        "engineering": {
            "anthropic/claude-fable-5-1": {
                "accuracy": 90.815,
                "latency": 66.832,
                "stderr": 0.928,
                "cost_per_test": 0.303892,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-fable-5": {
                "accuracy": 89.99,
                "latency": 80.54,
                "stderr": 0.964,
                "cost_per_test": 0.189415,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-5": {
                "accuracy": 89.886,
                "latency": 24.005,
                "stderr": 0.99,
                "cost_per_test": 0.059264,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.7-flash": {
                "accuracy": 89.267,
                "latency": 7.272,
                "stderr": 1.003,
                "cost_per_test": 0.023229,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.8-flash": {
                "accuracy": 88.751,
                "latency": 14.793,
                "stderr": 1.019,
                "cost_per_test": 0.048753,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "meta/muse_spark_1_1": {
                "accuracy": 88.132,
                "latency": 49.792,
                "stderr": 1.043,
                "cost_per_test": 0.024833,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 88.029,
                "latency": 64.579,
                "stderr": 1.043,
                "cost_per_test": 0.079443,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-4-7": {
                "accuracy": 88.029,
                "latency": 319.67,
                "stderr": 1.091,
                "cost_per_test": 0.248968,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.5-flash": {
                "accuracy": 87.822,
                "latency": 30.126,
                "stderr": 1.051,
                "cost_per_test": 0.061489,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.6-sol": {
                "accuracy": 87.822,
                "latency": 46.83,
                "stderr": 1.051,
                "cost_per_test": 0.096053,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-4-8": {
                "accuracy": 87.719,
                "latency": 72.117,
                "stderr": 1.054,
                "cost_per_test": 0.177437,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 87.616,
                "latency": 52.3,
                "stderr": 1.058,
                "cost_per_test": 0.080638,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 87.307,
                "latency": 66.298,
                "stderr": 1.069,
                "cost_per_test": 0.037611,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-4.5": {
                "accuracy": 87.203,
                "latency": 454.689,
                "stderr": 1.105,
                "cost_per_test": 0.040013,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 87.1,
                "latency": 225.755,
                "stderr": 1.159,
                "cost_per_test": 0.236159,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 87.1,
                "latency": 463.727,
                "stderr": 1.175,
                "cost_per_test": 0.053424,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "kimi/kimi-k3": {
                "accuracy": 86.894,
                "latency": 130.219,
                "stderr": 1.084,
                "cost_per_test": 0.072667,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "google/gemini-3.6-flash": {
                "accuracy": 86.687,
                "latency": 15.24,
                "stderr": 1.091,
                "cost_per_test": 0.029075,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3.8-max": {
                "accuracy": 86.687,
                "latency": 186.045,
                "stderr": 1.091,
                "cost_per_test": 0.054036,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 86.687,
                "latency": 877.3,
                "stderr": 1.139,
                "cost_per_test": 0.099767,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 86.481,
                "latency": 97.048,
                "stderr": 1.546,
                "cost_per_test": 0.575076,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "meta/muse_spark_1_2": {
                "accuracy": 86.275,
                "latency": 105.172,
                "stderr": 1.105,
                "cost_per_test": 0.029168,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "alibaba/qwen3.7-max": {
                "accuracy": 85.759,
                "latency": 61.801,
                "stderr": 1.123,
                "cost_per_test": 0.035533,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "grok/grok-4.6": {
                "accuracy": 85.759,
                "latency": 151.687,
                "stderr": 1.123,
                "cost_per_test": 0.038794,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "thinkingmachines/inkling-small": {
                "accuracy": 85.759,
                "latency": 206.532,
                "stderr": 1.123,
                "cost_per_test": 0.017356,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 262000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "openai/gpt-5.6-luna": {
                "accuracy": 85.655,
                "latency": 53.431,
                "stderr": 1.187,
                "cost_per_test": 0.009616,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "thinkingmachines/inkling": {
                "accuracy": 85.655,
                "latency": 206.526,
                "stderr": 1.126,
                "cost_per_test": 0.126553,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 85.243,
                "latency": 55.333,
                "stderr": 1.381,
                "cost_per_test": 0.123866,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3.5-plus-thinking": {
                "accuracy": 85.243,
                "latency": 152.966,
                "stderr": 1.139,
                "cost_per_test": 0.015775,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 85.243,
                "latency": 254.237,
                "stderr": 1.491,
                "cost_per_test": 0.309322,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 84.83,
                "latency": 79.599,
                "stderr": 1.152,
                "cost_per_test": 0.062984,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "meta/muse_spark": {
                "accuracy": 84.623,
                "latency": 79.232,
                "stderr": 1.159,
                "cost_per_test": 0.003009,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 84.314,
                "latency": 82.131,
                "stderr": 1.168,
                "cost_per_test": 0.016678,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "grok/grok-4.3": {
                "accuracy": 84.211,
                "latency": 426.7,
                "stderr": 1.171,
                "cost_per_test": 0.024835,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "zai/glm-5.1": {
                "accuracy": 84.029,
                "latency": 266.126,
                "stderr": 1.184,
                "cost_per_test": 0.049265,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "zai/glm-5.3": {
                "accuracy": 84.004,
                "latency": 175.352,
                "stderr": 1.178,
                "cost_per_test": 0.047363,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 83.798,
                "latency": 89.832,
                "stderr": 1.187,
                "cost_per_test": 0.018821,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 83.695,
                "latency": 14.796,
                "stderr": 1.187,
                "cost_per_test": 0.001948,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "zai/glm-5.2": {
                "accuracy": 83.385,
                "latency": 129.896,
                "stderr": 1.196,
                "cost_per_test": 0.037616,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "deepseek/deepseek-v4-pro-0813": {
                "accuracy": 83.385,
                "latency": 164.764,
                "stderr": 1.216,
                "cost_per_test": 0.042054,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "fireworks/deepseek-v3p2-thinking": {
                "accuracy": 83.282,
                "latency": 222.254,
                "stderr": 1.199,
                "cost_per_test": 0.003153,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 83.179,
                "latency": 32.25,
                "stderr": 1.202,
                "cost_per_test": 0.032227,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "deepseek/deepseek-v4-flash-0731": {
                "accuracy": 83.179,
                "latency": 52.692,
                "stderr": 1.202,
                "cost_per_test": 0.009277,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "zai/glm-5-thinking": {
                "accuracy": 83.179,
                "latency": 198.893,
                "stderr": 1.205,
                "cost_per_test": 0.030891,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 82.972,
                "latency": 116.232,
                "stderr": 1.21,
                "cost_per_test": 0.1194,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.6-terra": {
                "accuracy": 82.869,
                "latency": 26.937,
                "stderr": 1.21,
                "cost_per_test": 0.02277,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3.5-flash-lite": {
                "accuracy": 82.766,
                "latency": 9.671,
                "stderr": 1.213,
                "cost_per_test": 0.009929,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "nvidia/nemotron-3-ultra-550b-a55b": {
                "accuracy": 82.766,
                "latency": 132.631,
                "stderr": 1.213,
                "cost_per_test": null,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Nvidia",
                "harness": null
            },
            "alibaba/qwen3.5-flash": {
                "accuracy": 82.456,
                "latency": 56.165,
                "stderr": 1.222,
                "cost_per_test": 0.002943,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 82.456,
                "latency": 81.642,
                "stderr": 1.606,
                "cost_per_test": 0.093312,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-4-0709": {
                "accuracy": 82.456,
                "latency": 257.063,
                "stderr": 1.222,
                "cost_per_test": 0.147419,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "alibaba/qwen3.8-27b": {
                "accuracy": 82.353,
                "latency": 79.758,
                "stderr": 1.225,
                "cost_per_test": 0.035999,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 82.043,
                "latency": 51.562,
                "stderr": 1.239,
                "cost_per_test": 0.010098,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-4-1-20250805-thinking": {
                "accuracy": 82.043,
                "latency": 60.787,
                "stderr": 1.268,
                "cost_per_test": 0.265746,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "minimax/MiniMax-M2.1": {
                "accuracy": 81.837,
                "latency": 105.718,
                "stderr": 1.239,
                "cost_per_test": 0.008321,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "anthropic/claude-sonnet-5": {
                "accuracy": 81.631,
                "latency": 19.936,
                "stderr": 1.474,
                "cost_per_test": 0.037502,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "zai/glm-5.3-flash": {
                "accuracy": 81.631,
                "latency": 196.997,
                "stderr": 1.244,
                "cost_per_test": 0.004642,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "alibaba/qwen3-max-2026-01-23": {
                "accuracy": 81.631,
                "latency": 238.952,
                "stderr": 1.244,
                "cost_per_test": 0.065445,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "minimax/MiniMax-M3": {
                "accuracy": 81.115,
                "latency": 136.739,
                "stderr": 1.29,
                "cost_per_test": 0.017722,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 512000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "xiaomi/mimo-v2.5-pro": {
                "accuracy": 81.115,
                "latency": 217.255,
                "stderr": 1.375,
                "cost_per_test": 0.00665,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "xiaomi/mimo-v2.5": {
                "accuracy": 81.011,
                "latency": 77.878,
                "stderr": 1.263,
                "cost_per_test": 0.00212,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 80.599,
                "latency": 98.624,
                "stderr": 1.27,
                "cost_per_test": 0.062582,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 80.495,
                "latency": 60.928,
                "stderr": 1.278,
                "cost_per_test": 0.058109,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3-max": {
                "accuracy": 80.289,
                "latency": 0.0,
                "stderr": 1.278,
                "cost_per_test": 0.014173,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "fireworks/deepseek-r1": {
                "accuracy": 80.289,
                "latency": 63.374,
                "stderr": 1.52,
                "cost_per_test": 0.027306,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "grok/grok-4-fast-reasoning": {
                "accuracy": 80.083,
                "latency": 22.173,
                "stderr": 1.283,
                "cost_per_test": 0.00358,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-2.5-pro-exp-03-25": {
                "accuracy": 79.773,
                "latency": 38.297,
                "stderr": 1.29,
                "cost_per_test": 0.013937,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 79.773,
                "latency": 71.805,
                "stderr": 1.29,
                "cost_per_test": 0.003066,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "fireworks/qwen3-235b-a22b": {
                "accuracy": 79.154,
                "latency": 110.064,
                "stderr": 1.31,
                "cost_per_test": 0.005318,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "openai/o3-2025-04-16": {
                "accuracy": 78.638,
                "latency": 47.699,
                "stderr": 1.317,
                "cost_per_test": 0.029288,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-4.7": {
                "accuracy": 78.638,
                "latency": 291.489,
                "stderr": 1.317,
                "cost_per_test": 0.020993,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "mistralai/mistral-medium-3.5": {
                "accuracy": 78.225,
                "latency": 81.498,
                "stderr": 1.326,
                "cost_per_test": 0.086574,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "anthropic/claude-opus-4-1-20250805": {
                "accuracy": 77.709,
                "latency": 13.94,
                "stderr": 1.354,
                "cost_per_test": 0.073918,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "fireworks/deepseek-v3p2": {
                "accuracy": 77.606,
                "latency": 47.81,
                "stderr": 1.346,
                "cost_per_test": 0.000647,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "ant/ling-3.0-flash-2607": {
                "accuracy": 77.503,
                "latency": 106.872,
                "stderr": 1.356,
                "cost_per_test": 0.004868,
                "temperature": 0.6,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514-thinking": {
                "accuracy": 77.296,
                "latency": 98.218,
                "stderr": 1.346,
                "cost_per_test": 0.101515,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025-thinking": {
                "accuracy": 77.193,
                "latency": 20.916,
                "stderr": 1.348,
                "cost_per_test": 0.002718,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "zai/glm-4.6": {
                "accuracy": 77.193,
                "latency": 105.568,
                "stderr": 1.354,
                "cost_per_test": 0.015562,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 122880,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "alibaba/qwen3-max-preview": {
                "accuracy": 76.987,
                "latency": 104.441,
                "stderr": 1.352,
                "cost_per_test": 0.013442,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025": {
                "accuracy": 76.471,
                "latency": 16.513,
                "stderr": 1.363,
                "cost_per_test": 0.00273,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-4-20250514": {
                "accuracy": 76.367,
                "latency": 14.014,
                "stderr": 1.365,
                "cost_per_test": 0.067022,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-3-mini-fast-high-reasoning": {
                "accuracy": 76.367,
                "latency": 47.029,
                "stderr": 1.365,
                "cost_per_test": 0.004227,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101": {
                "accuracy": 76.161,
                "latency": 11.153,
                "stderr": 1.543,
                "cost_per_test": 0.072202,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "minimax/MiniMax-M2.5": {
                "accuracy": 75.439,
                "latency": 142.331,
                "stderr": 1.392,
                "cost_per_test": 0.015702,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "mistralai/mistral-large-2512": {
                "accuracy": 74.923,
                "latency": 26.843,
                "stderr": 1.595,
                "cost_per_test": 0.002489,
                "temperature": 0.065,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 74.613,
                "latency": 89.219,
                "stderr": 1.398,
                "cost_per_test": 0.013803,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 74.613,
                "latency": 146.95,
                "stderr": 1.415,
                "cost_per_test": 0.012338,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "openai/o1-2024-12-17": {
                "accuracy": 74.2,
                "latency": 62.165,
                "stderr": 1.406,
                "cost_per_test": 0.232772,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-3-7-sonnet-20250219-thinking": {
                "accuracy": 73.787,
                "latency": 67.263,
                "stderr": 1.413,
                "cost_per_test": 0.087638,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 73.271,
                "latency": 26.91,
                "stderr": 1.465,
                "cost_per_test": 0.001399,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "fireworks/deepseek-v3-0324": {
                "accuracy": 72.446,
                "latency": 67.828,
                "stderr": 1.465,
                "cost_per_test": 0.001946,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "openai/o4-mini-2025-04-16": {
                "accuracy": 72.136,
                "latency": 24.109,
                "stderr": 1.442,
                "cost_per_test": 0.010718,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-3-mini-fast-low-reasoning": {
                "accuracy": 71.93,
                "latency": 13.897,
                "stderr": 1.443,
                "cost_per_test": 0.001665,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "low",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 71.723,
                "latency": 59.966,
                "stderr": 1.57,
                "cost_per_test": 0.044915,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "fireworks/gpt-oss-120b": {
                "accuracy": 71.517,
                "latency": 49.348,
                "stderr": 1.455,
                "cost_per_test": 0.004726,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025-thinking": {
                "accuracy": 71.414,
                "latency": 20.77,
                "stderr": 1.451,
                "cost_per_test": 0.00048,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "fireworks/llama4-maverick-instruct-basic": {
                "accuracy": 70.898,
                "latency": 7.294,
                "stderr": 1.459,
                "cost_per_test": 0.000875,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514": {
                "accuracy": 70.898,
                "latency": 12.865,
                "stderr": 1.459,
                "cost_per_test": 0.013775,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "together/moonshotai/Kimi-K2-Instruct": {
                "accuracy": 69.453,
                "latency": 44.845,
                "stderr": 1.48,
                "cost_per_test": 0.004529,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "kimi/kimi-k2-thinking": {
                "accuracy": 69.35,
                "latency": 169.443,
                "stderr": 1.516,
                "cost_per_test": 0.003996,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "grok/grok-3": {
                "accuracy": 69.143,
                "latency": 16.967,
                "stderr": 1.484,
                "cost_per_test": 0.023894,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/o3-mini-2025-01-31": {
                "accuracy": 68.627,
                "latency": 50.797,
                "stderr": 1.491,
                "cost_per_test": 0.029435,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025": {
                "accuracy": 68.524,
                "latency": 6.686,
                "stderr": 1.492,
                "cost_per_test": 0.001175,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-4.1-2025-04-14": {
                "accuracy": 67.802,
                "latency": 16.506,
                "stderr": 1.501,
                "cost_per_test": 0.011514,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-2.0-flash-001": {
                "accuracy": 67.699,
                "latency": 9.302,
                "stderr": 1.502,
                "cost_per_test": 0.000771,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "zai/glm-4.5": {
                "accuracy": 67.389,
                "latency": 379.722,
                "stderr": 1.506,
                "cost_per_test": 0.021658,
                "temperature": 0.6,
                "top_p": 1.0,
                "max_output_tokens": 81920,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "anthropic/claude-3-7-sonnet-20250219": {
                "accuracy": 66.77,
                "latency": 8.402,
                "stderr": 1.513,
                "cost_per_test": 0.012613,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5-nano-2025-08-07": {
                "accuracy": 66.151,
                "latency": 52.506,
                "stderr": 1.52,
                "cost_per_test": 0.004676,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-4.1-mini-2025-04-14": {
                "accuracy": 65.325,
                "latency": 7.809,
                "stderr": 1.529,
                "cost_per_test": 0.002335,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-2-1212": {
                "accuracy": 62.229,
                "latency": 23.215,
                "stderr": 1.557,
                "cost_per_test": 0.010247,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 0,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "grok/grok-4-1-fast-non-reasoning": {
                "accuracy": 62.126,
                "latency": 6.197,
                "stderr": 1.558,
                "cost_per_test": 0.000363,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-3-5-sonnet-20241022": {
                "accuracy": 61.197,
                "latency": 6.929,
                "stderr": 1.565,
                "cost_per_test": 0.009441,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "together/meta-llama/Llama-4-Scout-17B-16E-Instruct": {
                "accuracy": 60.991,
                "latency": 7.111,
                "stderr": 1.567,
                "cost_per_test": 0.000583,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "fireworks/deepseek-v3": {
                "accuracy": 60.681,
                "latency": 28.367,
                "stderr": 1.571,
                "cost_per_test": 0.001626,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "google/gemini-1.5-pro-002": {
                "accuracy": 59.752,
                "latency": 7.355,
                "stderr": 1.575,
                "cost_per_test": 0.004022,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "mistralai/mistral-medium-2505": {
                "accuracy": 59.649,
                "latency": 20.497,
                "stderr": 1.581,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "grok/grok-4-fast-non-reasoning": {
                "accuracy": 58.308,
                "latency": 2.258,
                "stderr": 1.584,
                "cost_per_test": 0.001118,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "fireworks/gpt-oss-20b": {
                "accuracy": 56.45,
                "latency": 50.126,
                "stderr": 1.603,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "mistralai/magistral-medium-2509": {
                "accuracy": 56.45,
                "latency": 64.176,
                "stderr": 1.593,
                "cost_per_test": 0.030606,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "cohere/command-a-03-2025": {
                "accuracy": 55.521,
                "latency": 18.253,
                "stderr": 1.606,
                "cost_per_test": 0.011655,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 8000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "openai/gpt-4o-2024-08-06": {
                "accuracy": 55.108,
                "latency": 16.182,
                "stderr": 1.598,
                "cost_per_test": 0.010385,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "together/langston/nim/nvidia/llama-3.3-nemotron-super-49b-v1-42e84561-thinking": {
                "accuracy": 53.87,
                "latency": 83.21,
                "stderr": 1.603,
                "cost_per_test": null,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "poolside/laguna-xs.2": {
                "accuracy": 52.941,
                "latency": 122.193,
                "stderr": 1.596,
                "cost_per_test": 0.001904,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "together/meta-llama/Llama-3.3-70B-Instruct-Turbo": {
                "accuracy": 52.735,
                "latency": 5.892,
                "stderr": 1.604,
                "cost_per_test": 0.001797,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "mistralai/magistral-small-2509": {
                "accuracy": 52.735,
                "latency": 31.202,
                "stderr": 1.604,
                "cost_per_test": 0.00718,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "openai/gpt-4o-2024-11-20": {
                "accuracy": 52.632,
                "latency": 15.557,
                "stderr": 1.606,
                "cost_per_test": 0.010722,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "mistralai/mistral-large-2411": {
                "accuracy": 52.219,
                "latency": 16.566,
                "stderr": 1.605,
                "cost_per_test": 0.007721,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "together/langston/nim/nvidia/llama-3.3-nemotron-super-49b-v1-42e84561": {
                "accuracy": 52.219,
                "latency": 23.061,
                "stderr": 1.606,
                "cost_per_test": null,
                "temperature": 0.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "mistralai/mistral-small-2503": {
                "accuracy": 49.432,
                "latency": 7.592,
                "stderr": 1.606,
                "cost_per_test": 0.000322,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "poolside/laguna-m.1": {
                "accuracy": 49.123,
                "latency": 221.89,
                "stderr": 1.605,
                "cost_per_test": 0.005474,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "mistralai/mistral-small-2402": {
                "accuracy": 46.027,
                "latency": 9.576,
                "stderr": 1.601,
                "cost_per_test": 0.000792,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "anthropic/claude-3-5-haiku-20241022": {
                "accuracy": 45.924,
                "latency": 6.225,
                "stderr": 1.601,
                "cost_per_test": 0.002363,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-1.5-flash-002": {
                "accuracy": 44.788,
                "latency": 2.879,
                "stderr": 1.597,
                "cost_per_test": 0.000227,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-4o-mini-2024-07-18": {
                "accuracy": 39.216,
                "latency": 8.948,
                "stderr": 1.568,
                "cost_per_test": 0.000561,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-4.1-nano-2025-04-14": {
                "accuracy": 35.913,
                "latency": 5.105,
                "stderr": 1.527,
                "cost_per_test": 0.000638,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "ai21labs/jamba-large-1.6": {
                "accuracy": 31.579,
                "latency": 19.41,
                "stderr": 1.484,
                "cost_per_test": 0.007021,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "cohere/command-r-plus": {
                "accuracy": 26.109,
                "latency": 7.918,
                "stderr": 1.411,
                "cost_per_test": 0.00741,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 4000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "ai21labs/jamba-mini-1.6": {
                "accuracy": 17.441,
                "latency": 5.321,
                "stderr": 1.213,
                "cost_per_test": 0.000545,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            }
        },
        "health": {
            "anthropic/claude-opus-5": {
                "accuracy": 87.775,
                "latency": 8.814,
                "stderr": 1.145,
                "cost_per_test": 0.023061,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-fable-5-1": {
                "accuracy": 86.675,
                "latency": 18.621,
                "stderr": 1.188,
                "cost_per_test": 0.081746,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 86.43,
                "latency": 19.713,
                "stderr": 1.197,
                "cost_per_test": 0.018611,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-4-8": {
                "accuracy": 85.086,
                "latency": 17.113,
                "stderr": 1.246,
                "cost_per_test": 0.040556,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-4-1-20250805-thinking": {
                "accuracy": 84.963,
                "latency": 20.011,
                "stderr": 1.254,
                "cost_per_test": 0.085317,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.7-flash": {
                "accuracy": 84.719,
                "latency": 4.605,
                "stderr": 1.258,
                "cost_per_test": 0.008386,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.6-flash": {
                "accuracy": 84.597,
                "latency": 6.549,
                "stderr": 1.262,
                "cost_per_test": 0.010436,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-fable-5": {
                "accuracy": 84.597,
                "latency": 20.082,
                "stderr": 1.262,
                "cost_per_test": 0.084721,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 84.597,
                "latency": 25.958,
                "stderr": 1.603,
                "cost_per_test": 0.024715,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 84.597,
                "latency": 29.677,
                "stderr": 1.446,
                "cost_per_test": 0.15263,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.8-flash": {
                "accuracy": 84.474,
                "latency": 8.143,
                "stderr": 1.266,
                "cost_per_test": 0.019314,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3.7-max": {
                "accuracy": 84.352,
                "latency": 25.191,
                "stderr": 1.27,
                "cost_per_test": 0.013281,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 84.352,
                "latency": 35.101,
                "stderr": 1.27,
                "cost_per_test": 0.026769,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4.6": {
                "accuracy": 84.352,
                "latency": 38.646,
                "stderr": 1.27,
                "cost_per_test": 0.012673,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-5.6-sol": {
                "accuracy": 84.108,
                "latency": 11.833,
                "stderr": 1.278,
                "cost_per_test": 0.025299,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-4-1-20250805": {
                "accuracy": 84.108,
                "latency": 15.984,
                "stderr": 1.282,
                "cost_per_test": 0.015834,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101": {
                "accuracy": 83.985,
                "latency": 8.045,
                "stderr": 1.321,
                "cost_per_test": 0.04467,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-4-20250514": {
                "accuracy": 83.985,
                "latency": 9.711,
                "stderr": 1.282,
                "cost_per_test": 0.041684,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3.8-max": {
                "accuracy": 83.985,
                "latency": 61.797,
                "stderr": 1.282,
                "cost_per_test": 0.017996,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "grok/grok-4.5": {
                "accuracy": 83.985,
                "latency": 315.299,
                "stderr": 1.282,
                "cost_per_test": 0.011257,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "minimax/MiniMax-M2.1": {
                "accuracy": 83.741,
                "latency": 25.643,
                "stderr": 1.29,
                "cost_per_test": 0.001883,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "google/gemini-3.5-flash": {
                "accuracy": 83.619,
                "latency": 10.563,
                "stderr": 1.294,
                "cost_per_test": 0.019367,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 83.619,
                "latency": 22.817,
                "stderr": 1.294,
                "cost_per_test": 0.031299,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "meta/muse_spark_1_1": {
                "accuracy": 83.374,
                "latency": 19.769,
                "stderr": 1.309,
                "cost_per_test": 0.0099,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "openai/gpt-5.6-terra": {
                "accuracy": 82.763,
                "latency": 5.144,
                "stderr": 1.321,
                "cost_per_test": 0.006085,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 82.641,
                "latency": 37.102,
                "stderr": 1.324,
                "cost_per_test": 0.018444,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-4.3": {
                "accuracy": 82.641,
                "latency": 402.568,
                "stderr": 1.324,
                "cost_per_test": 0.00841,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 82.396,
                "latency": 29.104,
                "stderr": 1.553,
                "cost_per_test": 0.03146,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 82.396,
                "latency": 38.712,
                "stderr": 1.598,
                "cost_per_test": 0.045974,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-4-7": {
                "accuracy": 82.396,
                "latency": 78.119,
                "stderr": 1.332,
                "cost_per_test": 0.022124,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 82.396,
                "latency": 678.193,
                "stderr": 1.342,
                "cost_per_test": 0.016856,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "anthropic/claude-sonnet-5": {
                "accuracy": 82.274,
                "latency": 40.906,
                "stderr": 1.692,
                "cost_per_test": 0.065913,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "kimi/kimi-k3": {
                "accuracy": 82.029,
                "latency": 19.978,
                "stderr": 1.342,
                "cost_per_test": 0.015709,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 82.029,
                "latency": 29.09,
                "stderr": 1.49,
                "cost_per_test": 0.027233,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "meta/muse_spark_1_2": {
                "accuracy": 82.029,
                "latency": 41.654,
                "stderr": 1.346,
                "cost_per_test": 0.010082,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "zai/glm-5.3": {
                "accuracy": 81.785,
                "latency": 41.946,
                "stderr": 1.363,
                "cost_per_test": 0.010657,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "grok/grok-4-0709": {
                "accuracy": 81.663,
                "latency": 34.976,
                "stderr": 1.353,
                "cost_per_test": 0.030109,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "zai/glm-5.1": {
                "accuracy": 81.518,
                "latency": 42.893,
                "stderr": 1.358,
                "cost_per_test": 0.006832,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "openai/gpt-5.6-luna": {
                "accuracy": 81.418,
                "latency": 12.775,
                "stderr": 1.363,
                "cost_per_test": 0.001906,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-5.2": {
                "accuracy": 81.418,
                "latency": 41.625,
                "stderr": 1.36,
                "cost_per_test": 0.010657,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "deepseek/deepseek-v4-pro-0813": {
                "accuracy": 81.296,
                "latency": 53.62,
                "stderr": 1.4,
                "cost_per_test": 0.010372,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 81.174,
                "latency": 21.417,
                "stderr": 1.367,
                "cost_per_test": 0.022852,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 80.929,
                "latency": 8.982,
                "stderr": 1.374,
                "cost_per_test": 0.010252,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 80.929,
                "latency": 14.211,
                "stderr": 1.374,
                "cost_per_test": 0.011522,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-5.3-flash": {
                "accuracy": 80.929,
                "latency": 42.236,
                "stderr": 1.393,
                "cost_per_test": 0.000983,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 80.807,
                "latency": 28.123,
                "stderr": 1.377,
                "cost_per_test": 0.017724,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 80.807,
                "latency": 32.16,
                "stderr": 1.377,
                "cost_per_test": 0.006303,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 80.807,
                "latency": 35.864,
                "stderr": 1.377,
                "cost_per_test": 0.00687,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "meta/muse_spark": {
                "accuracy": 80.807,
                "latency": 59.9,
                "stderr": 1.377,
                "cost_per_test": 0.001615,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514": {
                "accuracy": 80.685,
                "latency": 7.345,
                "stderr": 1.38,
                "cost_per_test": 0.008541,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/o3-2025-04-16": {
                "accuracy": 80.685,
                "latency": 9.436,
                "stderr": 1.38,
                "cost_per_test": 0.007691,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "deepseek/deepseek-v4-flash-0731": {
                "accuracy": 80.685,
                "latency": 13.266,
                "stderr": 1.38,
                "cost_per_test": 0.001767,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 80.685,
                "latency": 123.418,
                "stderr": 1.38,
                "cost_per_test": 0.013682,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "google/gemini-3.5-flash-lite": {
                "accuracy": 80.44,
                "latency": 3.872,
                "stderr": 1.387,
                "cost_per_test": 0.003157,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 80.318,
                "latency": 6.968,
                "stderr": 1.39,
                "cost_per_test": 0.000835,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514-thinking": {
                "accuracy": 80.318,
                "latency": 28.811,
                "stderr": 1.39,
                "cost_per_test": 0.026623,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "minimax/MiniMax-M3": {
                "accuracy": 80.196,
                "latency": 29.591,
                "stderr": 1.406,
                "cost_per_test": 0.002572,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 512000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "zai/glm-5-thinking": {
                "accuracy": 80.196,
                "latency": 82.493,
                "stderr": 1.393,
                "cost_per_test": 0.008348,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 80.073,
                "latency": 16.235,
                "stderr": 1.406,
                "cost_per_test": 0.002732,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3.5-plus-thinking": {
                "accuracy": 80.073,
                "latency": 65.695,
                "stderr": 1.397,
                "cost_per_test": 0.008763,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "google/gemini-2.5-pro-exp-03-25": {
                "accuracy": 79.951,
                "latency": 15.379,
                "stderr": 1.4,
                "cost_per_test": 0.007625,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/o1-2024-12-17": {
                "accuracy": 79.951,
                "latency": 23.197,
                "stderr": 1.4,
                "cost_per_test": 0.0985,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "thinkingmachines/inkling": {
                "accuracy": 79.829,
                "latency": 59.033,
                "stderr": 1.403,
                "cost_per_test": 0.029508,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "nvidia/nemotron-3-ultra-550b-a55b": {
                "accuracy": 79.462,
                "latency": 17.419,
                "stderr": 1.412,
                "cost_per_test": null,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Nvidia",
                "harness": null
            },
            "zai/glm-4.6": {
                "accuracy": 79.462,
                "latency": 32.202,
                "stderr": 1.412,
                "cost_per_test": 0.004877,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 122880,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "anthropic/claude-3-7-sonnet-20250219-thinking": {
                "accuracy": 79.34,
                "latency": 24.979,
                "stderr": 1.416,
                "cost_per_test": 0.028831,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 79.218,
                "latency": 15.775,
                "stderr": 1.419,
                "cost_per_test": 0.000726,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "xiaomi/mimo-v2.5-pro": {
                "accuracy": 79.218,
                "latency": 35.649,
                "stderr": 1.748,
                "cost_per_test": 0.000991,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "thinkingmachines/inkling-small": {
                "accuracy": 79.218,
                "latency": 57.829,
                "stderr": 1.419,
                "cost_per_test": 0.003965,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 262000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "ant/ling-3.0-flash-2607": {
                "accuracy": 78.973,
                "latency": 5.73,
                "stderr": 1.428,
                "cost_per_test": 0.000228,
                "temperature": 0.6,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "anthropic/claude-3-5-sonnet-20241022": {
                "accuracy": 78.973,
                "latency": 6.52,
                "stderr": 1.425,
                "cost_per_test": 0.007583,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3-max-preview": {
                "accuracy": 78.851,
                "latency": 36.405,
                "stderr": 1.428,
                "cost_per_test": 0.00458,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "fireworks/deepseek-v3p2": {
                "accuracy": 78.851,
                "latency": 41.573,
                "stderr": 1.431,
                "cost_per_test": 0.000192,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "alibaba/qwen3-max-2026-01-23": {
                "accuracy": 78.729,
                "latency": 99.036,
                "stderr": 1.431,
                "cost_per_test": 0.022536,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "alibaba/qwen3.8-27b": {
                "accuracy": 78.606,
                "latency": 102.332,
                "stderr": 1.434,
                "cost_per_test": 0.007673,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "alibaba/qwen3-max": {
                "accuracy": 78.484,
                "latency": 0.0,
                "stderr": 1.437,
                "cost_per_test": 0.004609,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 78.484,
                "latency": 10.916,
                "stderr": 1.437,
                "cost_per_test": 0.003669,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3.5-flash": {
                "accuracy": 78.484,
                "latency": 30.761,
                "stderr": 1.437,
                "cost_per_test": 0.001703,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "anthropic/claude-3-7-sonnet-20250219": {
                "accuracy": 78.362,
                "latency": 6.152,
                "stderr": 1.44,
                "cost_per_test": 0.008436,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-3": {
                "accuracy": 78.362,
                "latency": 8.855,
                "stderr": 1.44,
                "cost_per_test": 0.013128,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "xiaomi/mimo-v2.5": {
                "accuracy": 78.362,
                "latency": 17.843,
                "stderr": 1.443,
                "cost_per_test": 0.000458,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "fireworks/deepseek-v3p2-thinking": {
                "accuracy": 78.362,
                "latency": 67.657,
                "stderr": 1.44,
                "cost_per_test": 0.000779,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "zai/glm-4.5": {
                "accuracy": 78.117,
                "latency": 57.607,
                "stderr": 1.446,
                "cost_per_test": 0.006468,
                "temperature": 0.6,
                "top_p": 1.0,
                "max_output_tokens": 81920,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "openai/gpt-4.1-2025-04-14": {
                "accuracy": 77.995,
                "latency": 2.974,
                "stderr": 1.448,
                "cost_per_test": 0.004239,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/o3-mini-2025-01-31": {
                "accuracy": 77.751,
                "latency": 18.661,
                "stderr": 1.454,
                "cost_per_test": 0.009501,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025": {
                "accuracy": 77.628,
                "latency": 4.928,
                "stderr": 1.457,
                "cost_per_test": 0.001017,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "zai/glm-4.7": {
                "accuracy": 77.628,
                "latency": 63.083,
                "stderr": 1.457,
                "cost_per_test": 0.004571,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "together/moonshotai/Kimi-K2-Instruct": {
                "accuracy": 77.506,
                "latency": 15.73,
                "stderr": 1.46,
                "cost_per_test": 0.002245,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "minimax/MiniMax-M2.5": {
                "accuracy": 77.506,
                "latency": 36.157,
                "stderr": 1.46,
                "cost_per_test": 0.003324,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 77.384,
                "latency": 29.718,
                "stderr": 1.476,
                "cost_per_test": 0.002225,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "mistralai/mistral-large-2512": {
                "accuracy": 77.262,
                "latency": 11.318,
                "stderr": 1.738,
                "cost_per_test": 0.00113,
                "temperature": 0.065,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025-thinking": {
                "accuracy": 77.139,
                "latency": 8.366,
                "stderr": 1.468,
                "cost_per_test": 0.001022,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "fireworks/deepseek-r1": {
                "accuracy": 77.139,
                "latency": 19.731,
                "stderr": 1.586,
                "cost_per_test": 0.008846,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "kimi/kimi-k2-thinking": {
                "accuracy": 76.773,
                "latency": 77.672,
                "stderr": 1.492,
                "cost_per_test": 0.001618,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "fireworks/deepseek-v3-0324": {
                "accuracy": 76.65,
                "latency": 22.284,
                "stderr": 1.479,
                "cost_per_test": 0.001227,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "openai/gpt-4o-2024-08-06": {
                "accuracy": 76.528,
                "latency": 5.636,
                "stderr": 1.482,
                "cost_per_test": 0.005575,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-3-mini-fast-high-reasoning": {
                "accuracy": 76.406,
                "latency": 8.07,
                "stderr": 1.485,
                "cost_per_test": 0.001042,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/o4-mini-2025-04-16": {
                "accuracy": 76.406,
                "latency": 8.799,
                "stderr": 1.485,
                "cost_per_test": 0.004377,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "fireworks/llama4-maverick-instruct-basic": {
                "accuracy": 76.284,
                "latency": 5.289,
                "stderr": 1.487,
                "cost_per_test": 0.000695,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "google/gemini-2.0-flash-001": {
                "accuracy": 75.917,
                "latency": 2.117,
                "stderr": 1.495,
                "cost_per_test": 0.000209,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "fireworks/gpt-oss-120b": {
                "accuracy": 75.795,
                "latency": 40.606,
                "stderr": 1.498,
                "cost_per_test": 0.001217,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "openai/gpt-4.1-mini-2025-04-14": {
                "accuracy": 75.672,
                "latency": 2.805,
                "stderr": 1.5,
                "cost_per_test": 0.000979,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-4o-2024-11-20": {
                "accuracy": 75.306,
                "latency": 6.075,
                "stderr": 1.525,
                "cost_per_test": 0.005606,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "mistralai/mistral-medium-3.5": {
                "accuracy": 75.306,
                "latency": 57.996,
                "stderr": 1.508,
                "cost_per_test": 0.023976,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025": {
                "accuracy": 75.061,
                "latency": 2.435,
                "stderr": 1.513,
                "cost_per_test": 0.000374,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025-thinking": {
                "accuracy": 75.061,
                "latency": 7.426,
                "stderr": 1.513,
                "cost_per_test": 0.000241,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "fireworks/qwen3-235b-a22b": {
                "accuracy": 75.061,
                "latency": 31.091,
                "stderr": 1.513,
                "cost_per_test": 0.001536,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "mistralai/mistral-medium-2505": {
                "accuracy": 74.939,
                "latency": 4.505,
                "stderr": 1.57,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "grok/grok-3-mini-fast-low-reasoning": {
                "accuracy": 74.817,
                "latency": 5.44,
                "stderr": 1.518,
                "cost_per_test": 0.000791,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "low",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "mistralai/magistral-medium-2509": {
                "accuracy": 74.817,
                "latency": 16.969,
                "stderr": 1.518,
                "cost_per_test": 0.008671,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "grok/grok-4-fast-reasoning": {
                "accuracy": 74.572,
                "latency": 5.016,
                "stderr": 1.523,
                "cost_per_test": 0.001232,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "fireworks/deepseek-v3": {
                "accuracy": 74.572,
                "latency": 18.211,
                "stderr": 1.523,
                "cost_per_test": 0.001068,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 74.572,
                "latency": 19.554,
                "stderr": 1.743,
                "cost_per_test": 0.011481,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-2-1212": {
                "accuracy": 73.961,
                "latency": 3.946,
                "stderr": 1.534,
                "cost_per_test": 0.003919,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 0,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "together/langston/nim/nvidia/llama-3.3-nemotron-super-49b-v1-42e84561-thinking": {
                "accuracy": 73.839,
                "latency": 20.723,
                "stderr": 1.546,
                "cost_per_test": null,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "grok/grok-4-1-fast-non-reasoning": {
                "accuracy": 73.594,
                "latency": 1.964,
                "stderr": 1.541,
                "cost_per_test": 0.000149,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-1.5-pro-002": {
                "accuracy": 73.594,
                "latency": 2.18,
                "stderr": 1.544,
                "cost_per_test": 0.001843,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5-nano-2025-08-07": {
                "accuracy": 73.227,
                "latency": 30.193,
                "stderr": 1.548,
                "cost_per_test": 0.001306,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "poolside/laguna-m.1": {
                "accuracy": 73.227,
                "latency": 81.138,
                "stderr": 1.574,
                "cost_per_test": 0.000919,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "together/meta-llama/Llama-3.3-70B-Instruct-Turbo": {
                "accuracy": 73.105,
                "latency": 4.147,
                "stderr": 1.55,
                "cost_per_test": 0.001362,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "mistralai/mistral-large-2411": {
                "accuracy": 72.861,
                "latency": 4.163,
                "stderr": 1.555,
                "cost_per_test": 0.003581,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 72.127,
                "latency": 5.585,
                "stderr": 1.674,
                "cost_per_test": 0.000201,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "together/meta-llama/Llama-4-Scout-17B-16E-Instruct": {
                "accuracy": 71.883,
                "latency": 4.621,
                "stderr": 1.572,
                "cost_per_test": 0.000437,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "mistralai/mistral-small-2503": {
                "accuracy": 70.782,
                "latency": 2.018,
                "stderr": 1.59,
                "cost_per_test": 0.000134,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "cohere/command-a-03-2025": {
                "accuracy": 70.538,
                "latency": 6.736,
                "stderr": 1.683,
                "cost_per_test": 0.006182,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 8000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "openai/gpt-4.1-nano-2025-04-14": {
                "accuracy": 69.682,
                "latency": 1.938,
                "stderr": 1.626,
                "cost_per_test": 0.000255,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "mistralai/mistral-small-2402": {
                "accuracy": 69.56,
                "latency": 2.709,
                "stderr": 1.609,
                "cost_per_test": 0.000342,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "poolside/laguna-xs.2": {
                "accuracy": 68.826,
                "latency": 32.333,
                "stderr": 1.745,
                "cost_per_test": 0.000352,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "fireworks/gpt-oss-20b": {
                "accuracy": 68.46,
                "latency": 45.25,
                "stderr": 1.673,
                "cost_per_test": 0.000695,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "together/langston/nim/nvidia/llama-3.3-nemotron-super-49b-v1-42e84561": {
                "accuracy": 67.848,
                "latency": 11.844,
                "stderr": 1.709,
                "cost_per_test": null,
                "temperature": 0.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "openai/gpt-4o-mini-2024-07-18": {
                "accuracy": 67.482,
                "latency": 3.66,
                "stderr": 1.638,
                "cost_per_test": 0.000298,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4-fast-non-reasoning": {
                "accuracy": 67.237,
                "latency": 1.761,
                "stderr": 1.641,
                "cost_per_test": 0.00089,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-3-5-haiku-20241022": {
                "accuracy": 66.748,
                "latency": 5.851,
                "stderr": 1.647,
                "cost_per_test": 0.001951,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-1.5-flash-002": {
                "accuracy": 62.714,
                "latency": 1.396,
                "stderr": 1.691,
                "cost_per_test": 0.000135,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "ai21labs/jamba-large-1.6": {
                "accuracy": 56.846,
                "latency": 7.652,
                "stderr": 1.732,
                "cost_per_test": 0.003323,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "mistralai/magistral-small-2509": {
                "accuracy": 54.279,
                "latency": 8.939,
                "stderr": 1.742,
                "cost_per_test": 0.002333,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "cohere/command-r-plus": {
                "accuracy": 46.088,
                "latency": 5.687,
                "stderr": 1.743,
                "cost_per_test": 0.005504,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 4000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "ai21labs/jamba-mini-1.6": {
                "accuracy": 43.399,
                "latency": 1.08,
                "stderr": 1.733,
                "cost_per_test": 0.000264,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            }
        },
        "history": {
            "anthropic/claude-opus-5": {
                "accuracy": 86.614,
                "latency": 8.395,
                "stderr": 1.744,
                "cost_per_test": 0.03216,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 85.564,
                "latency": 15.026,
                "stderr": 1.801,
                "cost_per_test": 0.018593,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-fable-5-1": {
                "accuracy": 85.302,
                "latency": 17.932,
                "stderr": 1.814,
                "cost_per_test": 0.099496,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-fable-5": {
                "accuracy": 85.302,
                "latency": 32.513,
                "stderr": 1.814,
                "cost_per_test": 0.079154,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 84.777,
                "latency": 19.767,
                "stderr": 1.84,
                "cost_per_test": 0.029725,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.7-flash": {
                "accuracy": 83.99,
                "latency": 3.589,
                "stderr": 1.879,
                "cost_per_test": 0.00988,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.8-flash": {
                "accuracy": 83.727,
                "latency": 7.452,
                "stderr": 1.891,
                "cost_per_test": 0.020687,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.6-flash": {
                "accuracy": 83.465,
                "latency": 5.946,
                "stderr": 1.903,
                "cost_per_test": 0.011323,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-4-7": {
                "accuracy": 83.202,
                "latency": 134.626,
                "stderr": 1.962,
                "cost_per_test": 0.04407,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-4-8": {
                "accuracy": 82.677,
                "latency": 12.47,
                "stderr": 1.939,
                "cost_per_test": 0.04551,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 82.677,
                "latency": 20.649,
                "stderr": 2.553,
                "cost_per_test": 0.022339,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-4-1-20250805": {
                "accuracy": 82.415,
                "latency": 20.301,
                "stderr": 1.95,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3.7-max": {
                "accuracy": 82.152,
                "latency": 24.783,
                "stderr": 1.962,
                "cost_per_test": 0.015035,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/gpt-5.6-sol": {
                "accuracy": 81.89,
                "latency": 9.19,
                "stderr": 1.973,
                "cost_per_test": 0.021669,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-4-1-20250805-thinking": {
                "accuracy": 81.89,
                "latency": 21.033,
                "stderr": 1.973,
                "cost_per_test": 0.102264,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 81.89,
                "latency": 29.478,
                "stderr": 2.159,
                "cost_per_test": 0.029215,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "meta/muse_spark_1_2": {
                "accuracy": 81.89,
                "latency": 60.188,
                "stderr": 1.973,
                "cost_per_test": 0.008557,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "google/gemini-3.5-flash": {
                "accuracy": 81.627,
                "latency": 11.228,
                "stderr": 1.984,
                "cost_per_test": 0.021196,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 81.627,
                "latency": 58.094,
                "stderr": 1.984,
                "cost_per_test": 0.025591,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4.5": {
                "accuracy": 81.365,
                "latency": 427.645,
                "stderr": 1.995,
                "cost_per_test": 0.011201,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 81.102,
                "latency": 24.579,
                "stderr": 2.282,
                "cost_per_test": 0.030658,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 81.102,
                "latency": 38.219,
                "stderr": 2.006,
                "cost_per_test": 0.007206,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "alibaba/qwen3.8-max": {
                "accuracy": 81.102,
                "latency": 40.614,
                "stderr": 2.006,
                "cost_per_test": 0.013065,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "grok/grok-4.6": {
                "accuracy": 81.102,
                "latency": 47.917,
                "stderr": 2.006,
                "cost_per_test": 0.01487,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 80.84,
                "latency": 27.069,
                "stderr": 2.016,
                "cost_per_test": 0.013377,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "meta/muse_spark": {
                "accuracy": 80.84,
                "latency": 47.223,
                "stderr": 2.016,
                "cost_per_test": 0.001956,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "anthropic/claude-opus-4-20250514": {
                "accuracy": 80.315,
                "latency": 11.317,
                "stderr": 2.047,
                "cost_per_test": 0.059706,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 80.315,
                "latency": 23.848,
                "stderr": 2.115,
                "cost_per_test": 0.132189,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-sonnet-5": {
                "accuracy": 80.315,
                "latency": 24.241,
                "stderr": 2.077,
                "cost_per_test": 0.046377,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "meta/muse_spark_1_1": {
                "accuracy": 79.79,
                "latency": 16.33,
                "stderr": 2.057,
                "cost_per_test": 0.00874,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "openai/gpt-5.6-terra": {
                "accuracy": 79.528,
                "latency": 10.208,
                "stderr": 2.067,
                "cost_per_test": 0.006033,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "minimax/MiniMax-M2.1": {
                "accuracy": 79.528,
                "latency": 38.904,
                "stderr": 2.067,
                "cost_per_test": 0.003043,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514": {
                "accuracy": 79.265,
                "latency": 10.566,
                "stderr": 2.077,
                "cost_per_test": 0.012016,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "kimi/kimi-k3": {
                "accuracy": 79.265,
                "latency": 14.649,
                "stderr": 2.077,
                "cost_per_test": 0.01165,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 79.265,
                "latency": 20.393,
                "stderr": 2.4,
                "cost_per_test": 0.040285,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 79.265,
                "latency": 23.049,
                "stderr": 2.077,
                "cost_per_test": 0.01705,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3.5-plus-thinking": {
                "accuracy": 78.74,
                "latency": 75.191,
                "stderr": 2.096,
                "cost_per_test": 0.008337,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "google/gemini-3.5-flash-lite": {
                "accuracy": 78.478,
                "latency": 3.845,
                "stderr": 2.105,
                "cost_per_test": 0.003345,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101": {
                "accuracy": 78.478,
                "latency": 8.614,
                "stderr": 2.124,
                "cost_per_test": 0.061382,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "deepseek/deepseek-v4-pro-0813": {
                "accuracy": 78.478,
                "latency": 36.299,
                "stderr": 2.115,
                "cost_per_test": 0.006593,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "zai/glm-5.3": {
                "accuracy": 78.478,
                "latency": 38.643,
                "stderr": 2.105,
                "cost_per_test": 0.010747,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 78.478,
                "latency": 96.434,
                "stderr": 2.105,
                "cost_per_test": 0.011629,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "zai/glm-5.1": {
                "accuracy": 78.421,
                "latency": 32.724,
                "stderr": 2.11,
                "cost_per_test": 0.006012,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 78.215,
                "latency": 36.902,
                "stderr": 2.115,
                "cost_per_test": 0.000893,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/o3-2025-04-16": {
                "accuracy": 77.953,
                "latency": 9.482,
                "stderr": 2.124,
                "cost_per_test": 0.009669,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4.3": {
                "accuracy": 77.953,
                "latency": 3277.635,
                "stderr": 2.124,
                "cost_per_test": 0.007354,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 77.69,
                "latency": 22.006,
                "stderr": 2.133,
                "cost_per_test": 0.009824,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 77.69,
                "latency": 635.878,
                "stderr": 2.193,
                "cost_per_test": 0.010836,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 77.428,
                "latency": 7.596,
                "stderr": 2.142,
                "cost_per_test": 0.010326,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "zai/glm-5-thinking": {
                "accuracy": 77.428,
                "latency": 57.957,
                "stderr": 2.142,
                "cost_per_test": 0.007885,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "anthropic/claude-3-7-sonnet-20250219": {
                "accuracy": 77.165,
                "latency": 6.155,
                "stderr": 2.151,
                "cost_per_test": 0.011353,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "deepseek/deepseek-v4-flash-0731": {
                "accuracy": 77.165,
                "latency": 10.586,
                "stderr": 2.151,
                "cost_per_test": 0.001394,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "zai/glm-5.2": {
                "accuracy": 76.903,
                "latency": 38.522,
                "stderr": 2.159,
                "cost_per_test": 0.010321,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "thinkingmachines/inkling-small": {
                "accuracy": 76.903,
                "latency": 47.346,
                "stderr": 2.159,
                "cost_per_test": 0.003762,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 262000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "xiaomi/mimo-v2.5-pro": {
                "accuracy": 76.64,
                "latency": 39.88,
                "stderr": 2.526,
                "cost_per_test": 0.001114,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025-thinking": {
                "accuracy": 76.378,
                "latency": 9.309,
                "stderr": 2.176,
                "cost_per_test": 0.001366,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-2.5-pro-exp-03-25": {
                "accuracy": 76.378,
                "latency": 15.675,
                "stderr": 2.176,
                "cost_per_test": 0.009111,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 76.378,
                "latency": 21.72,
                "stderr": 2.201,
                "cost_per_test": 0.020553,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 76.378,
                "latency": 27.199,
                "stderr": 2.176,
                "cost_per_test": 0.005502,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "nvidia/nemotron-3-ultra-550b-a55b": {
                "accuracy": 76.115,
                "latency": 13.577,
                "stderr": 2.184,
                "cost_per_test": null,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Nvidia",
                "harness": null
            },
            "together/langston/nim/nvidia/llama-3.3-nemotron-super-49b-v1-42e84561": {
                "accuracy": 76.115,
                "latency": 21.861,
                "stderr": 2.216,
                "cost_per_test": null,
                "temperature": 0.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "thinkingmachines/inkling": {
                "accuracy": 76.115,
                "latency": 43.959,
                "stderr": 2.184,
                "cost_per_test": 0.023192,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "fireworks/deepseek-v3p2-thinking": {
                "accuracy": 76.115,
                "latency": 61.056,
                "stderr": 2.184,
                "cost_per_test": 0.000701,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "openai/gpt-5.6-luna": {
                "accuracy": 75.853,
                "latency": 10.163,
                "stderr": 2.209,
                "cost_per_test": 0.001759,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 75.853,
                "latency": 15.211,
                "stderr": 2.193,
                "cost_per_test": 0.000752,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "alibaba/qwen3.8-27b": {
                "accuracy": 75.591,
                "latency": 18.797,
                "stderr": 2.201,
                "cost_per_test": 0.006689,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514-thinking": {
                "accuracy": 75.591,
                "latency": 31.974,
                "stderr": 2.201,
                "cost_per_test": 0.029222,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-4-0709": {
                "accuracy": 75.591,
                "latency": 50.819,
                "stderr": 2.201,
                "cost_per_test": 0.032609,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-3-5-sonnet-20241022": {
                "accuracy": 75.328,
                "latency": 6.513,
                "stderr": 2.209,
                "cost_per_test": 0.010784,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 75.328,
                "latency": 15.341,
                "stderr": 2.209,
                "cost_per_test": 0.002869,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "fireworks/deepseek-v3p2": {
                "accuracy": 75.328,
                "latency": 63.49,
                "stderr": 2.216,
                "cost_per_test": 0.000199,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "openai/o1-2024-12-17": {
                "accuracy": 75.066,
                "latency": 23.108,
                "stderr": 2.216,
                "cost_per_test": 0.103124,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025": {
                "accuracy": 74.803,
                "latency": 5.53,
                "stderr": 2.224,
                "cost_per_test": 0.001343,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "zai/glm-5.3-flash": {
                "accuracy": 74.803,
                "latency": 21.322,
                "stderr": 2.224,
                "cost_per_test": 0.000572,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "alibaba/qwen3-max-2026-01-23": {
                "accuracy": 74.803,
                "latency": 89.177,
                "stderr": 2.224,
                "cost_per_test": 0.02017,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "alibaba/qwen3-max": {
                "accuracy": 74.278,
                "latency": 0.0,
                "stderr": 2.239,
                "cost_per_test": 0.005804,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "anthropic/claude-3-7-sonnet-20250219-thinking": {
                "accuracy": 74.278,
                "latency": 19.362,
                "stderr": 2.239,
                "cost_per_test": 0.024612,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-4-fast-reasoning": {
                "accuracy": 74.016,
                "latency": 4.846,
                "stderr": 2.247,
                "cost_per_test": 0.001969,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "fireworks/deepseek-r1": {
                "accuracy": 74.016,
                "latency": 16.728,
                "stderr": 2.385,
                "cost_per_test": 0.009046,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "alibaba/qwen3-max-preview": {
                "accuracy": 73.491,
                "latency": 30.172,
                "stderr": 2.261,
                "cost_per_test": 0.005582,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "minimax/MiniMax-M3": {
                "accuracy": 73.228,
                "latency": 25.596,
                "stderr": 2.268,
                "cost_per_test": 0.00265,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 512000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "zai/glm-4.7": {
                "accuracy": 72.703,
                "latency": 61.031,
                "stderr": 2.282,
                "cost_per_test": 0.005314,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "openai/gpt-4.1-2025-04-14": {
                "accuracy": 72.178,
                "latency": 3.257,
                "stderr": 2.296,
                "cost_per_test": 0.006572,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "kimi/kimi-k2-thinking": {
                "accuracy": 72.178,
                "latency": 87.832,
                "stderr": 2.302,
                "cost_per_test": 0.002134,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "openai/gpt-4o-2024-08-06": {
                "accuracy": 71.916,
                "latency": 6.494,
                "stderr": 2.302,
                "cost_per_test": 0.007865,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 71.654,
                "latency": 14.023,
                "stderr": 2.309,
                "cost_per_test": 0.003525,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "fireworks/qwen3-235b-a22b": {
                "accuracy": 71.654,
                "latency": 37.78,
                "stderr": 2.322,
                "cost_per_test": 0.001624,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "alibaba/qwen3.5-flash": {
                "accuracy": 71.391,
                "latency": 36.51,
                "stderr": 2.315,
                "cost_per_test": 0.001704,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "zai/glm-4.5": {
                "accuracy": 71.129,
                "latency": 64.589,
                "stderr": 2.322,
                "cost_per_test": 0.00669,
                "temperature": 0.6,
                "top_p": 1.0,
                "max_output_tokens": 81920,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "grok/grok-2-1212": {
                "accuracy": 70.866,
                "latency": 3.751,
                "stderr": 2.328,
                "cost_per_test": 0.006042,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 0,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/o4-mini-2025-04-16": {
                "accuracy": 70.866,
                "latency": 7.315,
                "stderr": 2.328,
                "cost_per_test": 0.005467,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "mistralai/mistral-large-2512": {
                "accuracy": 70.866,
                "latency": 11.778,
                "stderr": 2.555,
                "cost_per_test": 0.001663,
                "temperature": 0.065,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "grok/grok-3": {
                "accuracy": 70.604,
                "latency": 11.052,
                "stderr": 2.334,
                "cost_per_test": 0.017408,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 70.341,
                "latency": 18.203,
                "stderr": 2.561,
                "cost_per_test": 0.012087,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "fireworks/deepseek-v3-0324": {
                "accuracy": 70.079,
                "latency": 7.615,
                "stderr": 2.346,
                "cost_per_test": 0.002016,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "grok/grok-3-mini-fast-high-reasoning": {
                "accuracy": 69.816,
                "latency": 7.998,
                "stderr": 2.352,
                "cost_per_test": 0.001357,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "fireworks/llama4-maverick-instruct-basic": {
                "accuracy": 69.816,
                "latency": 28.434,
                "stderr": 2.352,
                "cost_per_test": 0.000734,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "ant/ling-3.0-flash-2607": {
                "accuracy": 69.554,
                "latency": 4.636,
                "stderr": 2.358,
                "cost_per_test": 0.000237,
                "temperature": 0.6,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "mistralai/mistral-medium-3.5": {
                "accuracy": 69.554,
                "latency": 52.52,
                "stderr": 2.358,
                "cost_per_test": 0.030174,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "together/moonshotai/Kimi-K2-Instruct": {
                "accuracy": 69.291,
                "latency": 14.335,
                "stderr": 2.363,
                "cost_per_test": 0.003214,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "google/gemini-2.0-flash-001": {
                "accuracy": 69.029,
                "latency": 1.28,
                "stderr": 2.369,
                "cost_per_test": 0.000268,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-3-mini-fast-low-reasoning": {
                "accuracy": 69.029,
                "latency": 5.387,
                "stderr": 2.369,
                "cost_per_test": 0.001109,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "low",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "xiaomi/mimo-v2.5": {
                "accuracy": 68.241,
                "latency": 16.687,
                "stderr": 2.39,
                "cost_per_test": 0.000427,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "fireworks/gpt-oss-120b": {
                "accuracy": 68.241,
                "latency": 29.697,
                "stderr": 2.385,
                "cost_per_test": 0.001333,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "fireworks/deepseek-v3": {
                "accuracy": 67.979,
                "latency": 2.571,
                "stderr": 2.39,
                "cost_per_test": 0.001981,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025": {
                "accuracy": 67.979,
                "latency": 2.792,
                "stderr": 2.39,
                "cost_per_test": 0.000524,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "mistralai/mistral-medium-2505": {
                "accuracy": 67.979,
                "latency": 3.949,
                "stderr": 2.405,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "zai/glm-4.6": {
                "accuracy": 67.979,
                "latency": 36.461,
                "stderr": 2.39,
                "cost_per_test": 0.005921,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 122880,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "minimax/MiniMax-M2.5": {
                "accuracy": 67.454,
                "latency": 34.015,
                "stderr": 2.4,
                "cost_per_test": 0.004019,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "google/gemini-1.5-pro-002": {
                "accuracy": 67.192,
                "latency": 1.967,
                "stderr": 2.405,
                "cost_per_test": 0.003125,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-4.1-mini-2025-04-14": {
                "accuracy": 67.192,
                "latency": 3.02,
                "stderr": 2.405,
                "cost_per_test": 0.001395,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025-thinking": {
                "accuracy": 66.404,
                "latency": 8.115,
                "stderr": 2.42,
                "cost_per_test": 0.000345,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-4-1-fast-non-reasoning": {
                "accuracy": 66.142,
                "latency": 2.078,
                "stderr": 2.424,
                "cost_per_test": 0.000239,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/o3-mini-2025-01-31": {
                "accuracy": 65.879,
                "latency": 20.203,
                "stderr": 2.429,
                "cost_per_test": 0.012645,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-4o-2024-11-20": {
                "accuracy": 65.617,
                "latency": 5.575,
                "stderr": 2.54,
                "cost_per_test": 0.008195,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 65.617,
                "latency": 29.914,
                "stderr": 2.438,
                "cost_per_test": 0.002667,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "together/meta-llama/Llama-3.3-70B-Instruct-Turbo": {
                "accuracy": 65.092,
                "latency": 1.88,
                "stderr": 2.442,
                "cost_per_test": 0.00206,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "mistralai/magistral-medium-2509": {
                "accuracy": 65.092,
                "latency": 18.999,
                "stderr": 2.442,
                "cost_per_test": 0.011342,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "together/langston/nim/nvidia/llama-3.3-nemotron-super-49b-v1-42e84561-thinking": {
                "accuracy": 64.829,
                "latency": 19.588,
                "stderr": 2.446,
                "cost_per_test": null,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "mistralai/mistral-large-2411": {
                "accuracy": 64.042,
                "latency": 3.039,
                "stderr": 2.458,
                "cost_per_test": 0.005579,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "cohere/command-a-03-2025": {
                "accuracy": 64.042,
                "latency": 5.903,
                "stderr": 2.474,
                "cost_per_test": 0.009169,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 8000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "anthropic/claude-3-5-haiku-20241022": {
                "accuracy": 63.255,
                "latency": 5.903,
                "stderr": 2.47,
                "cost_per_test": 0.00281,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-4-fast-non-reasoning": {
                "accuracy": 63.089,
                "latency": 2.238,
                "stderr": 2.469,
                "cost_per_test": 0.001623,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-5-nano-2025-08-07": {
                "accuracy": 61.942,
                "latency": 38.253,
                "stderr": 2.487,
                "cost_per_test": 0.001643,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "fireworks/gpt-oss-20b": {
                "accuracy": 61.68,
                "latency": 18.353,
                "stderr": 2.5,
                "cost_per_test": 0.000833,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "poolside/laguna-m.1": {
                "accuracy": 61.68,
                "latency": 27.185,
                "stderr": 2.524,
                "cost_per_test": 0.000802,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "poolside/laguna-xs.2": {
                "accuracy": 60.367,
                "latency": 30.011,
                "stderr": 2.561,
                "cost_per_test": 0.000367,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "mistralai/mistral-small-2503": {
                "accuracy": 59.58,
                "latency": 1.878,
                "stderr": 2.514,
                "cost_per_test": 0.00021,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 59.318,
                "latency": 4.698,
                "stderr": 2.54,
                "cost_per_test": 0.000241,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-1.5-flash-002": {
                "accuracy": 58.53,
                "latency": 1.045,
                "stderr": 2.524,
                "cost_per_test": 0.000199,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "together/meta-llama/Llama-4-Scout-17B-16E-Instruct": {
                "accuracy": 58.53,
                "latency": 5.104,
                "stderr": 2.524,
                "cost_per_test": 0.000675,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "mistralai/magistral-small-2509": {
                "accuracy": 58.005,
                "latency": 8.725,
                "stderr": 2.529,
                "cost_per_test": 0.002683,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "openai/gpt-4.1-nano-2025-04-14": {
                "accuracy": 55.381,
                "latency": 1.88,
                "stderr": 2.547,
                "cost_per_test": 0.000367,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-4o-mini-2024-07-18": {
                "accuracy": 54.593,
                "latency": 2.482,
                "stderr": 2.551,
                "cost_per_test": 0.00042,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "mistralai/mistral-small-2402": {
                "accuracy": 54.593,
                "latency": 3.661,
                "stderr": 2.551,
                "cost_per_test": 0.000569,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "cohere/command-r-plus": {
                "accuracy": 49.344,
                "latency": 1.934,
                "stderr": 2.561,
                "cost_per_test": 0.00626,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 4000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "ai21labs/jamba-mini-1.6": {
                "accuracy": 34.121,
                "latency": 1.15,
                "stderr": 2.429,
                "cost_per_test": 0.000471,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "ai21labs/jamba-large-1.6": {
                "accuracy": 28.871,
                "latency": 3.054,
                "stderr": 2.322,
                "cost_per_test": 0.004932,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            }
        },
        "law": {
            "anthropic/claude-fable-5-1": {
                "accuracy": 87.466,
                "latency": 22.156,
                "stderr": 0.998,
                "cost_per_test": 0.105418,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-5": {
                "accuracy": 86.558,
                "latency": 11.668,
                "stderr": 1.028,
                "cost_per_test": 0.032255,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-fable-5": {
                "accuracy": 86.467,
                "latency": 23.395,
                "stderr": 1.031,
                "cost_per_test": 0.104874,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 84.105,
                "latency": 22.32,
                "stderr": 1.102,
                "cost_per_test": 0.025774,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 82.47,
                "latency": 30.789,
                "stderr": 1.146,
                "cost_per_test": 0.044563,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-4-7": {
                "accuracy": 82.107,
                "latency": 259.071,
                "stderr": 1.184,
                "cost_per_test": 0.098415,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.8-flash": {
                "accuracy": 82.016,
                "latency": 10.668,
                "stderr": 1.162,
                "cost_per_test": 0.030686,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-4.5": {
                "accuracy": 82.016,
                "latency": 394.222,
                "stderr": 1.175,
                "cost_per_test": 0.023265,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-3.7-flash": {
                "accuracy": 81.926,
                "latency": 4.67,
                "stderr": 1.16,
                "cost_per_test": 0.011639,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-4-8": {
                "accuracy": 81.926,
                "latency": 46.643,
                "stderr": 1.16,
                "cost_per_test": 0.113953,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-4.6": {
                "accuracy": 81.835,
                "latency": 108.139,
                "stderr": 1.162,
                "cost_per_test": 0.022452,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-3.5-flash": {
                "accuracy": 80.381,
                "latency": 45.82,
                "stderr": 1.199,
                "cost_per_test": 0.028158,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.6-flash": {
                "accuracy": 80.109,
                "latency": 8.787,
                "stderr": 1.203,
                "cost_per_test": 0.014604,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-4-1-20250805-thinking": {
                "accuracy": 79.927,
                "latency": 26.755,
                "stderr": 1.209,
                "cost_per_test": 0.111326,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 79.746,
                "latency": 60.609,
                "stderr": 1.506,
                "cost_per_test": 0.055171,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3.7-max": {
                "accuracy": 79.473,
                "latency": 32.261,
                "stderr": 1.217,
                "cost_per_test": 0.017397,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/gpt-5.6-sol": {
                "accuracy": 79.11,
                "latency": 25.942,
                "stderr": 1.225,
                "cost_per_test": 0.053742,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "minimax/MiniMax-M2.1": {
                "accuracy": 78.928,
                "latency": 43.414,
                "stderr": 1.229,
                "cost_per_test": 0.003182,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "anthropic/claude-sonnet-5": {
                "accuracy": 78.747,
                "latency": 6.521,
                "stderr": 1.266,
                "cost_per_test": 0.015722,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 78.656,
                "latency": 66.38,
                "stderr": 1.342,
                "cost_per_test": 0.312744,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 78.474,
                "latency": 54.017,
                "stderr": 1.239,
                "cost_per_test": 0.027027,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101": {
                "accuracy": 78.292,
                "latency": 10.208,
                "stderr": 1.254,
                "cost_per_test": 0.054909,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-4-1-20250805": {
                "accuracy": 77.475,
                "latency": 18.284,
                "stderr": 1.259,
                "cost_per_test": 0.009339,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3.8-max": {
                "accuracy": 77.293,
                "latency": 136.785,
                "stderr": 1.263,
                "cost_per_test": 0.035829,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "kimi/kimi-k3": {
                "accuracy": 77.021,
                "latency": 52.421,
                "stderr": 1.268,
                "cost_per_test": 0.026007,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 76.93,
                "latency": 68.6,
                "stderr": 1.27,
                "cost_per_test": 0.053088,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "meta/muse_spark_1_2": {
                "accuracy": 76.658,
                "latency": 63.765,
                "stderr": 1.28,
                "cost_per_test": 0.014682,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "meta/muse_spark_1_1": {
                "accuracy": 76.476,
                "latency": 25.461,
                "stderr": 1.288,
                "cost_per_test": 0.012445,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 75.84,
                "latency": 33.183,
                "stderr": 1.29,
                "cost_per_test": 0.024716,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 75.658,
                "latency": 49.942,
                "stderr": 1.293,
                "cost_per_test": 0.009164,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 75.477,
                "latency": 45.188,
                "stderr": 1.456,
                "cost_per_test": 0.093825,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 75.295,
                "latency": 54.392,
                "stderr": 1.3,
                "cost_per_test": 0.034743,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 75.023,
                "latency": 10.251,
                "stderr": 1.305,
                "cost_per_test": 0.001055,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "meta/muse_spark": {
                "accuracy": 75.023,
                "latency": 47.886,
                "stderr": 1.305,
                "cost_per_test": 0.001906,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 74.932,
                "latency": 75.03,
                "stderr": 1.463,
                "cost_per_test": 0.063558,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "deepseek/deepseek-v4-pro-0813": {
                "accuracy": 74.932,
                "latency": 118.275,
                "stderr": 1.312,
                "cost_per_test": 0.022229,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "anthropic/claude-opus-4-20250514": {
                "accuracy": 74.569,
                "latency": 11.27,
                "stderr": 1.312,
                "cost_per_test": 0.050234,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.6-terra": {
                "accuracy": 74.387,
                "latency": 9.223,
                "stderr": 1.317,
                "cost_per_test": 0.011381,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4.3": {
                "accuracy": 74.387,
                "latency": 345.632,
                "stderr": 1.315,
                "cost_per_test": 0.013321,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 74.296,
                "latency": 272.45,
                "stderr": 1.323,
                "cost_per_test": 0.028894,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "zai/glm-5.3": {
                "accuracy": 74.205,
                "latency": 119.076,
                "stderr": 1.319,
                "cost_per_test": 0.032863,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 74.024,
                "latency": 535.644,
                "stderr": 1.345,
                "cost_per_test": 0.02826,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "zai/glm-5.1": {
                "accuracy": 74.0,
                "latency": 57.748,
                "stderr": 1.323,
                "cost_per_test": 0.010337,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "alibaba/qwen3.5-plus-thinking": {
                "accuracy": 73.933,
                "latency": 98.798,
                "stderr": 1.323,
                "cost_per_test": 0.011867,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 73.569,
                "latency": 72.778,
                "stderr": 1.442,
                "cost_per_test": 0.073486,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 73.479,
                "latency": 72.826,
                "stderr": 1.35,
                "cost_per_test": 0.067295,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/o3-2025-04-16": {
                "accuracy": 73.297,
                "latency": 20.864,
                "stderr": 1.333,
                "cost_per_test": 0.012843,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.6-luna": {
                "accuracy": 73.025,
                "latency": 29.858,
                "stderr": 1.339,
                "cost_per_test": 0.005721,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 72.934,
                "latency": 13.977,
                "stderr": 1.339,
                "cost_per_test": 0.016795,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "nvidia/nemotron-3-ultra-550b-a55b": {
                "accuracy": 72.934,
                "latency": 33.366,
                "stderr": 1.339,
                "cost_per_test": null,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Nvidia",
                "harness": null
            },
            "zai/glm-5-thinking": {
                "accuracy": 72.843,
                "latency": 97.959,
                "stderr": 1.34,
                "cost_per_test": 0.012339,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "google/gemini-3.5-flash-lite": {
                "accuracy": 72.48,
                "latency": 6.529,
                "stderr": 1.346,
                "cost_per_test": 0.005367,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "deepseek/deepseek-v4-flash-0731": {
                "accuracy": 72.207,
                "latency": 27.801,
                "stderr": 1.35,
                "cost_per_test": 0.003767,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "zai/glm-5.2": {
                "accuracy": 72.025,
                "latency": 72.715,
                "stderr": 1.353,
                "cost_per_test": 0.018016,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "fireworks/deepseek-v3p2-thinking": {
                "accuracy": 72.025,
                "latency": 105.186,
                "stderr": 1.353,
                "cost_per_test": 0.001281,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "openai/o1-2024-12-17": {
                "accuracy": 71.571,
                "latency": 29.349,
                "stderr": 1.359,
                "cost_per_test": 0.137459,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-5.3-flash": {
                "accuracy": 71.299,
                "latency": 65.411,
                "stderr": 1.363,
                "cost_per_test": 0.001518,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "thinkingmachines/inkling": {
                "accuracy": 71.208,
                "latency": 162.72,
                "stderr": 1.365,
                "cost_per_test": 0.087214,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 71.117,
                "latency": 42.667,
                "stderr": 1.366,
                "cost_per_test": 0.001114,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-2.5-pro-exp-03-25": {
                "accuracy": 71.026,
                "latency": 18.956,
                "stderr": 1.367,
                "cost_per_test": 0.009806,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-4-0709": {
                "accuracy": 71.026,
                "latency": 150.025,
                "stderr": 1.367,
                "cost_per_test": 0.058366,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "alibaba/qwen3-max-2026-01-23": {
                "accuracy": 70.481,
                "latency": 141.377,
                "stderr": 1.375,
                "cost_per_test": 0.032196,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "thinkingmachines/inkling-small": {
                "accuracy": 70.391,
                "latency": 132.291,
                "stderr": 1.376,
                "cost_per_test": 0.013026,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 262000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 70.3,
                "latency": 50.75,
                "stderr": 1.377,
                "cost_per_test": 0.009989,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 70.118,
                "latency": 30.578,
                "stderr": 1.384,
                "cost_per_test": 0.006384,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025": {
                "accuracy": 69.846,
                "latency": 29.965,
                "stderr": 1.383,
                "cost_per_test": 0.001465,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3-max": {
                "accuracy": 69.755,
                "latency": 0.0,
                "stderr": 1.384,
                "cost_per_test": 0.00685,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514": {
                "accuracy": 69.664,
                "latency": 11.73,
                "stderr": 1.385,
                "cost_per_test": 0.010381,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025-thinking": {
                "accuracy": 68.756,
                "latency": 11.802,
                "stderr": 1.397,
                "cost_per_test": 0.001468,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "fireworks/deepseek-r1": {
                "accuracy": 68.756,
                "latency": 30.476,
                "stderr": 1.476,
                "cost_per_test": 0.013571,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "grok/grok-4-fast-reasoning": {
                "accuracy": 68.483,
                "latency": 8.045,
                "stderr": 1.4,
                "cost_per_test": 0.001762,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "fireworks/deepseek-v3p2": {
                "accuracy": 68.211,
                "latency": 38.848,
                "stderr": 1.415,
                "cost_per_test": 0.000331,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "alibaba/qwen3-max-preview": {
                "accuracy": 68.029,
                "latency": 44.386,
                "stderr": 1.406,
                "cost_per_test": 0.006265,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "alibaba/qwen3.8-27b": {
                "accuracy": 67.847,
                "latency": 54.787,
                "stderr": 1.408,
                "cost_per_test": 0.020667,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "anthropic/claude-3-5-sonnet-20241022": {
                "accuracy": 67.666,
                "latency": 6.734,
                "stderr": 1.41,
                "cost_per_test": 0.009054,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-3-7-sonnet-20250219-thinking": {
                "accuracy": 67.575,
                "latency": 32.762,
                "stderr": 1.411,
                "cost_per_test": 0.036944,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3.5-flash": {
                "accuracy": 66.939,
                "latency": 48.279,
                "stderr": 1.418,
                "cost_per_test": 0.002347,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "xiaomi/mimo-v2.5-pro": {
                "accuracy": 66.939,
                "latency": 74.224,
                "stderr": 1.507,
                "cost_per_test": 0.002031,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514-thinking": {
                "accuracy": 66.939,
                "latency": 81.708,
                "stderr": 1.418,
                "cost_per_test": 0.069952,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-3-7-sonnet-20250219": {
                "accuracy": 66.848,
                "latency": 6.963,
                "stderr": 1.419,
                "cost_per_test": 0.010008,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "xiaomi/mimo-v2.5": {
                "accuracy": 66.757,
                "latency": 37.066,
                "stderr": 1.42,
                "cost_per_test": 0.000908,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "minimax/MiniMax-M3": {
                "accuracy": 66.485,
                "latency": 57.733,
                "stderr": 1.429,
                "cost_per_test": 0.005721,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 512000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "openai/gpt-4.1-2025-04-14": {
                "accuracy": 65.577,
                "latency": 4.849,
                "stderr": 1.432,
                "cost_per_test": 0.006256,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-4.5": {
                "accuracy": 65.123,
                "latency": 98.1,
                "stderr": 1.436,
                "cost_per_test": 0.008511,
                "temperature": 0.6,
                "top_p": 1.0,
                "max_output_tokens": 81920,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "zai/glm-4.7": {
                "accuracy": 64.759,
                "latency": 121.12,
                "stderr": 1.44,
                "cost_per_test": 0.00951,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "together/langston/nim/nvidia/llama-3.3-nemotron-super-49b-v1-42e84561": {
                "accuracy": 64.124,
                "latency": 17.7,
                "stderr": 1.446,
                "cost_per_test": null,
                "temperature": 0.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "grok/grok-3-mini-fast-high-reasoning": {
                "accuracy": 63.669,
                "latency": 10.826,
                "stderr": 1.449,
                "cost_per_test": 0.001425,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "kimi/kimi-k2-thinking": {
                "accuracy": 63.306,
                "latency": 145.75,
                "stderr": 1.466,
                "cost_per_test": 0.003132,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 62.761,
                "latency": 22.692,
                "stderr": 1.457,
                "cost_per_test": 0.006819,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "together/moonshotai/Kimi-K2-Instruct": {
                "accuracy": 62.67,
                "latency": 21.29,
                "stderr": 1.458,
                "cost_per_test": 0.00311,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 62.216,
                "latency": 43.716,
                "stderr": 1.507,
                "cost_per_test": 0.025813,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/o4-mini-2025-04-16": {
                "accuracy": 61.399,
                "latency": 12.946,
                "stderr": 1.467,
                "cost_per_test": 0.006345,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-3-mini-fast-low-reasoning": {
                "accuracy": 60.854,
                "latency": 6.139,
                "stderr": 1.471,
                "cost_per_test": 0.000978,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "low",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "fireworks/llama4-maverick-instruct-basic": {
                "accuracy": 60.763,
                "latency": 3.779,
                "stderr": 1.472,
                "cost_per_test": 0.000633,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "zai/glm-4.6": {
                "accuracy": 60.763,
                "latency": 47.136,
                "stderr": 1.472,
                "cost_per_test": 0.00724,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 122880,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "google/gemini-2.0-flash-001": {
                "accuracy": 60.672,
                "latency": 5.07,
                "stderr": 1.472,
                "cost_per_test": 0.000399,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025": {
                "accuracy": 60.581,
                "latency": 3.583,
                "stderr": 1.473,
                "cost_per_test": 0.000563,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 60.49,
                "latency": 64.611,
                "stderr": 1.476,
                "cost_per_test": 0.004826,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "grok/grok-3": {
                "accuracy": 60.218,
                "latency": 11.861,
                "stderr": 1.475,
                "cost_per_test": 0.019033,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "grok/grok-4-1-fast-non-reasoning": {
                "accuracy": 59.582,
                "latency": 2.596,
                "stderr": 1.479,
                "cost_per_test": 0.000217,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "grok/grok-2-1212": {
                "accuracy": 59.401,
                "latency": 5.726,
                "stderr": 1.48,
                "cost_per_test": 0.00557,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 0,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "fireworks/qwen3-235b-a22b": {
                "accuracy": 59.401,
                "latency": 71.288,
                "stderr": 1.49,
                "cost_per_test": 0.002407,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "minimax/MiniMax-M2.5": {
                "accuracy": 59.219,
                "latency": 66.126,
                "stderr": 1.482,
                "cost_per_test": 0.006817,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025-thinking": {
                "accuracy": 59.037,
                "latency": 10.155,
                "stderr": 1.482,
                "cost_per_test": 0.000305,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "fireworks/gpt-oss-120b": {
                "accuracy": 59.037,
                "latency": 69.15,
                "stderr": 1.483,
                "cost_per_test": 0.002942,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 58.856,
                "latency": 7.031,
                "stderr": 1.493,
                "cost_per_test": 0.000369,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "ant/ling-3.0-flash-2607": {
                "accuracy": 58.492,
                "latency": 17.187,
                "stderr": 1.489,
                "cost_per_test": 0.000711,
                "temperature": 0.6,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "grok/grok-4-fast-non-reasoning": {
                "accuracy": 57.856,
                "latency": 1.884,
                "stderr": 1.488,
                "cost_per_test": 0.001188,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "mistralai/mistral-medium-3.5": {
                "accuracy": 56.857,
                "latency": 72.657,
                "stderr": 1.493,
                "cost_per_test": 0.053547,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "mistralai/magistral-medium-2509": {
                "accuracy": 56.494,
                "latency": 37.236,
                "stderr": 1.494,
                "cost_per_test": 0.018636,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "mistralai/mistral-large-2512": {
                "accuracy": 56.403,
                "latency": 17.014,
                "stderr": 1.306,
                "cost_per_test": 0.001604,
                "temperature": 0.065,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "fireworks/deepseek-v3-0324": {
                "accuracy": 56.403,
                "latency": 29.997,
                "stderr": 1.494,
                "cost_per_test": 0.001441,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "openai/gpt-4o-2024-08-06": {
                "accuracy": 56.131,
                "latency": 8.064,
                "stderr": 1.496,
                "cost_per_test": 0.00696,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "together/langston/nim/nvidia/llama-3.3-nemotron-super-49b-v1-42e84561-thinking": {
                "accuracy": 55.223,
                "latency": 24.625,
                "stderr": 1.499,
                "cost_per_test": null,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "google/gemini-1.5-pro-002": {
                "accuracy": 54.223,
                "latency": 2.483,
                "stderr": 1.501,
                "cost_per_test": 0.002383,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-4.1-mini-2025-04-14": {
                "accuracy": 53.769,
                "latency": 3.66,
                "stderr": 1.503,
                "cost_per_test": 0.001316,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "mistralai/mistral-medium-2505": {
                "accuracy": 52.77,
                "latency": 6.141,
                "stderr": 1.505,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "poolside/laguna-xs.2": {
                "accuracy": 52.407,
                "latency": 34.147,
                "stderr": 1.461,
                "cost_per_test": 0.000428,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "poolside/laguna-m.1": {
                "accuracy": 52.044,
                "latency": 94.539,
                "stderr": 1.507,
                "cost_per_test": 0.00127,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "together/meta-llama/Llama-3.3-70B-Instruct-Turbo": {
                "accuracy": 51.771,
                "latency": 4.355,
                "stderr": 1.506,
                "cost_per_test": 0.001661,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "cohere/command-a-03-2025": {
                "accuracy": 51.408,
                "latency": 8.478,
                "stderr": 1.506,
                "cost_per_test": 0.008141,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 8000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "fireworks/deepseek-v3": {
                "accuracy": 51.045,
                "latency": 14.726,
                "stderr": 1.507,
                "cost_per_test": 0.001382,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "openai/o3-mini-2025-01-31": {
                "accuracy": 50.772,
                "latency": 31.103,
                "stderr": 1.507,
                "cost_per_test": 0.018975,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5-nano-2025-08-07": {
                "accuracy": 50.5,
                "latency": 32.578,
                "stderr": 1.507,
                "cost_per_test": 0.003021,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "mistralai/mistral-large-2411": {
                "accuracy": 49.046,
                "latency": 4.608,
                "stderr": 1.507,
                "cost_per_test": 0.004517,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "openai/gpt-4o-2024-11-20": {
                "accuracy": 48.683,
                "latency": 13.023,
                "stderr": 1.481,
                "cost_per_test": 0.008599,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "together/meta-llama/Llama-4-Scout-17B-16E-Instruct": {
                "accuracy": 47.048,
                "latency": 5.019,
                "stderr": 1.504,
                "cost_per_test": 0.00053,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "mistralai/mistral-small-2503": {
                "accuracy": 45.504,
                "latency": 2.897,
                "stderr": 1.501,
                "cost_per_test": 0.000181,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "fireworks/gpt-oss-20b": {
                "accuracy": 44.596,
                "latency": 73.861,
                "stderr": 1.484,
                "cost_per_test": 0.001907,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "anthropic/claude-3-5-haiku-20241022": {
                "accuracy": 44.142,
                "latency": 6.261,
                "stderr": 1.496,
                "cost_per_test": 0.002355,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-1.5-flash-002": {
                "accuracy": 42.598,
                "latency": 1.323,
                "stderr": 1.49,
                "cost_per_test": 0.000161,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "mistralai/mistral-small-2402": {
                "accuracy": 40.872,
                "latency": 3.275,
                "stderr": 1.482,
                "cost_per_test": 0.000464,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "openai/gpt-4o-mini-2024-07-18": {
                "accuracy": 37.693,
                "latency": 3.593,
                "stderr": 1.461,
                "cost_per_test": 0.000363,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-4.1-nano-2025-04-14": {
                "accuracy": 37.511,
                "latency": 2.468,
                "stderr": 1.456,
                "cost_per_test": 0.000364,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "cohere/command-r-plus": {
                "accuracy": 36.058,
                "latency": 3.06,
                "stderr": 1.446,
                "cost_per_test": 0.005053,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 4000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "ai21labs/jamba-large-1.6": {
                "accuracy": 35.695,
                "latency": 7.047,
                "stderr": 1.443,
                "cost_per_test": 0.004165,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "mistralai/magistral-small-2509": {
                "accuracy": 28.247,
                "latency": 17.029,
                "stderr": 1.357,
                "cost_per_test": 0.003685,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "ai21labs/jamba-mini-1.6": {
                "accuracy": 27.157,
                "latency": 1.471,
                "stderr": 1.34,
                "cost_per_test": 0.000355,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            }
        },
        "math": {
            "anthropic/claude-fable-5": {
                "accuracy": 97.557,
                "latency": 21.371,
                "stderr": 0.42,
                "cost_per_test": 0.07686,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-fable-5-1": {
                "accuracy": 97.335,
                "latency": 11.096,
                "stderr": 0.438,
                "cost_per_test": 0.065167,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-5": {
                "accuracy": 97.113,
                "latency": 7.774,
                "stderr": 0.461,
                "cost_per_test": 0.028227,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.5-flash": {
                "accuracy": 96.299,
                "latency": 50.407,
                "stderr": 0.533,
                "cost_per_test": 0.023637,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.8-flash": {
                "accuracy": 96.151,
                "latency": 5.444,
                "stderr": 0.523,
                "cost_per_test": 0.018302,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.7-flash": {
                "accuracy": 96.003,
                "latency": 3.581,
                "stderr": 0.538,
                "cost_per_test": 0.011336,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.6-flash": {
                "accuracy": 96.003,
                "latency": 6.718,
                "stderr": 0.533,
                "cost_per_test": 0.013897,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "meta/muse_spark_1_1": {
                "accuracy": 96.003,
                "latency": 18.126,
                "stderr": 0.547,
                "cost_per_test": 0.009214,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 96.003,
                "latency": 24.058,
                "stderr": 0.533,
                "cost_per_test": 0.03927,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 96.003,
                "latency": 24.464,
                "stderr": 0.533,
                "cost_per_test": 0.026993,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3.7-max": {
                "accuracy": 96.003,
                "latency": 29.213,
                "stderr": 0.533,
                "cost_per_test": 0.019017,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "meta/muse_spark_1_2": {
                "accuracy": 96.003,
                "latency": 31.834,
                "stderr": 0.533,
                "cost_per_test": 0.008477,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "anthropic/claude-sonnet-5": {
                "accuracy": 95.929,
                "latency": 36.687,
                "stderr": 0.685,
                "cost_per_test": 0.068593,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-4.6": {
                "accuracy": 95.929,
                "latency": 39.766,
                "stderr": 0.538,
                "cost_per_test": 0.013407,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 95.929,
                "latency": 41.961,
                "stderr": 1.024,
                "cost_per_test": 0.051479,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-4.5": {
                "accuracy": 95.929,
                "latency": 340.418,
                "stderr": 0.551,
                "cost_per_test": 0.014781,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-opus-4-8": {
                "accuracy": 95.855,
                "latency": 18.449,
                "stderr": 0.542,
                "cost_per_test": 0.055314,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "zai/glm-5.2": {
                "accuracy": 95.855,
                "latency": 40.377,
                "stderr": 0.542,
                "cost_per_test": 0.012165,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "anthropic/claude-opus-4-7": {
                "accuracy": 95.855,
                "latency": 79.094,
                "stderr": 0.542,
                "cost_per_test": 0.042614,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 95.707,
                "latency": 36.885,
                "stderr": 0.551,
                "cost_per_test": 0.007629,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 95.707,
                "latency": 39.099,
                "stderr": 0.551,
                "cost_per_test": 0.022316,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "kimi/kimi-k3": {
                "accuracy": 95.633,
                "latency": 33.852,
                "stderr": 0.556,
                "cost_per_test": 0.016185,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 95.633,
                "latency": 140.155,
                "stderr": 0.556,
                "cost_per_test": 0.014751,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "nvidia/nemotron-3-ultra-550b-a55b": {
                "accuracy": 95.559,
                "latency": 30.914,
                "stderr": 0.56,
                "cost_per_test": null,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Nvidia",
                "harness": null
            },
            "alibaba/qwen3.5-plus-thinking": {
                "accuracy": 95.559,
                "latency": 78.179,
                "stderr": 0.56,
                "cost_per_test": 0.009015,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "alibaba/qwen3-max-2026-01-23": {
                "accuracy": 95.559,
                "latency": 89.333,
                "stderr": 0.56,
                "cost_per_test": 0.024008,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "alibaba/qwen3.8-max": {
                "accuracy": 95.485,
                "latency": 41.346,
                "stderr": 0.565,
                "cost_per_test": 0.013284,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 95.485,
                "latency": 50.573,
                "stderr": 0.565,
                "cost_per_test": 0.00986,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "zai/glm-5.1": {
                "accuracy": 95.485,
                "latency": 62.824,
                "stderr": 0.565,
                "cost_per_test": 0.01127,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 95.337,
                "latency": 474.812,
                "stderr": 0.578,
                "cost_per_test": 0.022908,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "openai/gpt-5.6-sol": {
                "accuracy": 95.263,
                "latency": 9.448,
                "stderr": 0.578,
                "cost_per_test": 0.021345,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "thinkingmachines/inkling-small": {
                "accuracy": 95.263,
                "latency": 43.855,
                "stderr": 0.578,
                "cost_per_test": 0.00394,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 262000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "zai/glm-5.3": {
                "accuracy": 95.189,
                "latency": 34.75,
                "stderr": 0.582,
                "cost_per_test": 0.01228,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "zai/glm-5.3-flash": {
                "accuracy": 95.189,
                "latency": 35.793,
                "stderr": 0.582,
                "cost_per_test": 0.000859,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 95.115,
                "latency": 22.816,
                "stderr": 0.586,
                "cost_per_test": 0.013096,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "deepseek/deepseek-v4-pro-0813": {
                "accuracy": 95.115,
                "latency": 39.306,
                "stderr": 0.603,
                "cost_per_test": 0.008654,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "alibaba/qwen3-max": {
                "accuracy": 95.041,
                "latency": 0.0,
                "stderr": 0.591,
                "cost_per_test": 0.007666,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 95.041,
                "latency": 26.339,
                "stderr": 0.599,
                "cost_per_test": 0.031514,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "meta/muse_spark": {
                "accuracy": 95.041,
                "latency": 56.02,
                "stderr": 0.591,
                "cost_per_test": 0.001713,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "fireworks/deepseek-v3p2-thinking": {
                "accuracy": 95.041,
                "latency": 73.223,
                "stderr": 0.591,
                "cost_per_test": 0.000995,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 94.967,
                "latency": 37.162,
                "stderr": 1.123,
                "cost_per_test": 0.048284,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3.5-flash": {
                "accuracy": 94.893,
                "latency": 27.458,
                "stderr": 0.599,
                "cost_per_test": 0.001691,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "thinkingmachines/inkling": {
                "accuracy": 94.893,
                "latency": 46.611,
                "stderr": 0.599,
                "cost_per_test": 0.023878,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 94.819,
                "latency": 10.308,
                "stderr": 0.603,
                "cost_per_test": 0.001361,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "zai/glm-5-thinking": {
                "accuracy": 94.819,
                "latency": 79.691,
                "stderr": 0.603,
                "cost_per_test": 0.012724,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "alibaba/qwen3.8-27b": {
                "accuracy": 94.819,
                "latency": 97.073,
                "stderr": 0.603,
                "cost_per_test": 0.008576,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 94.745,
                "latency": 33.027,
                "stderr": 0.607,
                "cost_per_test": 0.017893,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 94.745,
                "latency": 36.702,
                "stderr": 0.607,
                "cost_per_test": 0.023739,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 94.671,
                "latency": 16.611,
                "stderr": 0.805,
                "cost_per_test": 0.037007,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-4-1-20250805": {
                "accuracy": 94.597,
                "latency": 19.462,
                "stderr": 0.619,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "minimax/MiniMax-M3": {
                "accuracy": 94.597,
                "latency": 34.289,
                "stderr": 0.675,
                "cost_per_test": 0.004775,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 512000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "ant/ling-3.0-flash-2607": {
                "accuracy": 94.523,
                "latency": 15.983,
                "stderr": 0.671,
                "cost_per_test": 0.00061,
                "temperature": 0.6,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "google/gemini-3.5-flash-lite": {
                "accuracy": 94.449,
                "latency": 4.909,
                "stderr": 0.623,
                "cost_per_test": 0.005287,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 94.449,
                "latency": 15.14,
                "stderr": 0.638,
                "cost_per_test": 0.00289,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 94.301,
                "latency": 12.822,
                "stderr": 0.631,
                "cost_per_test": 0.004542,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 94.301,
                "latency": 20.251,
                "stderr": 0.635,
                "cost_per_test": 0.019754,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3-max-preview": {
                "accuracy": 94.301,
                "latency": 44.587,
                "stderr": 0.631,
                "cost_per_test": 0.007185,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "anthropic/claude-opus-4-1-20250805-thinking": {
                "accuracy": 94.226,
                "latency": 36.869,
                "stderr": 0.653,
                "cost_per_test": 0.180541,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.6-terra": {
                "accuracy": 94.152,
                "latency": 5.296,
                "stderr": 0.638,
                "cost_per_test": 0.005209,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 94.152,
                "latency": 34.451,
                "stderr": 1.088,
                "cost_per_test": 0.228935,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 94.078,
                "latency": 45.334,
                "stderr": 1.353,
                "cost_per_test": 0.065662,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "deepseek/deepseek-v4-flash-0731": {
                "accuracy": 94.004,
                "latency": 12.347,
                "stderr": 0.646,
                "cost_per_test": 0.002039,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "openai/o3-2025-04-16": {
                "accuracy": 94.004,
                "latency": 20.683,
                "stderr": 0.646,
                "cost_per_test": 0.01422,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "xiaomi/mimo-v2.5": {
                "accuracy": 94.004,
                "latency": 27.882,
                "stderr": 0.653,
                "cost_per_test": 0.000807,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025-thinking": {
                "accuracy": 93.708,
                "latency": 10.711,
                "stderr": 0.661,
                "cost_per_test": 0.001851,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-4-20250514": {
                "accuracy": 93.708,
                "latency": 10.815,
                "stderr": 0.661,
                "cost_per_test": 0.064139,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "zai/glm-4.6": {
                "accuracy": 93.634,
                "latency": 51.314,
                "stderr": 0.699,
                "cost_per_test": 0.00669,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 122880,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "xiaomi/mimo-v2.5-pro": {
                "accuracy": 93.634,
                "latency": 71.077,
                "stderr": 0.86,
                "cost_per_test": 0.002403,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "openai/gpt-5.6-luna": {
                "accuracy": 93.412,
                "latency": 10.848,
                "stderr": 0.728,
                "cost_per_test": 0.001959,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "fireworks/gpt-oss-120b": {
                "accuracy": 93.338,
                "latency": 17.095,
                "stderr": 0.685,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "openai/o3-mini-2025-01-31": {
                "accuracy": 93.264,
                "latency": 20.432,
                "stderr": 0.682,
                "cost_per_test": 0.011498,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025": {
                "accuracy": 93.116,
                "latency": 7.385,
                "stderr": 0.689,
                "cost_per_test": 0.00183,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514-thinking": {
                "accuracy": 93.042,
                "latency": 64.992,
                "stderr": 0.696,
                "cost_per_test": 0.064348,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "fireworks/deepseek-r1": {
                "accuracy": 92.968,
                "latency": 26.179,
                "stderr": 1.043,
                "cost_per_test": 0.012333,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "zai/glm-4.7": {
                "accuracy": 92.968,
                "latency": 96.951,
                "stderr": 0.696,
                "cost_per_test": 0.008694,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "openai/gpt-5-nano-2025-08-07": {
                "accuracy": 92.82,
                "latency": 18.394,
                "stderr": 0.702,
                "cost_per_test": 0.001493,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "minimax/MiniMax-M2.1": {
                "accuracy": 92.746,
                "latency": 55.213,
                "stderr": 0.706,
                "cost_per_test": 0.004308,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 92.672,
                "latency": 49.349,
                "stderr": 0.712,
                "cost_per_test": 0.004255,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "minimax/MiniMax-M2.5": {
                "accuracy": 92.598,
                "latency": 42.988,
                "stderr": 0.719,
                "cost_per_test": 0.005288,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101": {
                "accuracy": 92.302,
                "latency": 7.848,
                "stderr": 1.063,
                "cost_per_test": 0.065281,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/o4-mini-2025-04-16": {
                "accuracy": 92.302,
                "latency": 9.219,
                "stderr": 0.725,
                "cost_per_test": 0.0055,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/o1-2024-12-17": {
                "accuracy": 91.932,
                "latency": 24.494,
                "stderr": 0.744,
                "cost_per_test": 0.12593,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "fireworks/deepseek-v3p2": {
                "accuracy": 91.71,
                "latency": 28.831,
                "stderr": 0.889,
                "cost_per_test": 0.000321,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "kimi/kimi-k2-thinking": {
                "accuracy": 91.636,
                "latency": 76.794,
                "stderr": 0.92,
                "cost_per_test": 0.002082,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "fireworks/qwen3-235b-a22b": {
                "accuracy": 91.562,
                "latency": 48.296,
                "stderr": 0.852,
                "cost_per_test": 0.002828,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "anthropic/claude-3-7-sonnet-20250219-thinking": {
                "accuracy": 91.414,
                "latency": 40.166,
                "stderr": 0.762,
                "cost_per_test": 0.05626,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-3-mini-fast-high-reasoning": {
                "accuracy": 91.266,
                "latency": 26.956,
                "stderr": 0.768,
                "cost_per_test": 0.002248,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "grok/grok-3": {
                "accuracy": 91.192,
                "latency": 14.214,
                "stderr": 0.771,
                "cost_per_test": 0.022041,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025-thinking": {
                "accuracy": 91.118,
                "latency": 9.519,
                "stderr": 0.774,
                "cost_per_test": 0.000398,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025": {
                "accuracy": 91.044,
                "latency": 3.386,
                "stderr": 0.777,
                "cost_per_test": 0.000703,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "fireworks/deepseek-v3-0324": {
                "accuracy": 91.044,
                "latency": 23.543,
                "stderr": 0.8,
                "cost_per_test": 0.001944,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "mistralai/mistral-medium-3.5": {
                "accuracy": 90.97,
                "latency": 48.241,
                "stderr": 0.78,
                "cost_per_test": 0.037224,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514": {
                "accuracy": 90.896,
                "latency": 9.118,
                "stderr": 0.783,
                "cost_per_test": 0.013312,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 90.896,
                "latency": 9.175,
                "stderr": 0.886,
                "cost_per_test": 0.000463,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4-0709": {
                "accuracy": 90.822,
                "latency": 57.184,
                "stderr": 0.786,
                "cost_per_test": 0.051539,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-4.1-mini-2025-04-14": {
                "accuracy": 90.674,
                "latency": 4.803,
                "stderr": 0.791,
                "cost_per_test": 0.001774,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-3-mini-fast-low-reasoning": {
                "accuracy": 90.6,
                "latency": 7.892,
                "stderr": 0.794,
                "cost_per_test": 0.001214,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "low",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 90.6,
                "latency": 29.009,
                "stderr": 0.794,
                "cost_per_test": 0.001272,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-4.1-2025-04-14": {
                "accuracy": 89.637,
                "latency": 8.747,
                "stderr": 0.829,
                "cost_per_test": 0.010057,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "fireworks/llama4-maverick-instruct-basic": {
                "accuracy": 89.563,
                "latency": 5.065,
                "stderr": 0.832,
                "cost_per_test": 0.000803,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "grok/grok-4-fast-reasoning": {
                "accuracy": 88.601,
                "latency": 7.219,
                "stderr": 0.865,
                "cost_per_test": 0.002105,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "mistralai/mistral-large-2512": {
                "accuracy": 88.379,
                "latency": 15.539,
                "stderr": 1.232,
                "cost_per_test": 0.001912,
                "temperature": 0.065,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "fireworks/gpt-oss-20b": {
                "accuracy": 87.935,
                "latency": 20.684,
                "stderr": 1.014,
                "cost_per_test": 0.000939,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "together/moonshotai/Kimi-K2-Instruct": {
                "accuracy": 87.935,
                "latency": 27.447,
                "stderr": 0.886,
                "cost_per_test": 0.003777,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "zai/glm-4.5": {
                "accuracy": 87.861,
                "latency": 115.062,
                "stderr": 0.889,
                "cost_per_test": 0.012745,
                "temperature": 0.6,
                "top_p": 1.0,
                "max_output_tokens": 81920,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "anthropic/claude-3-7-sonnet-20250219": {
                "accuracy": 86.158,
                "latency": 6.24,
                "stderr": 0.94,
                "cost_per_test": 0.012018,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-1.5-pro-002": {
                "accuracy": 85.936,
                "latency": 4.791,
                "stderr": 0.946,
                "cost_per_test": 0.003813,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "fireworks/deepseek-v3": {
                "accuracy": 84.826,
                "latency": 16.389,
                "stderr": 0.976,
                "cost_per_test": 0.001766,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 84.752,
                "latency": 23.459,
                "stderr": 1.217,
                "cost_per_test": 0.020505,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-2.5-pro-exp-03-25": {
                "accuracy": 83.79,
                "latency": 22.856,
                "stderr": 1.003,
                "cost_per_test": 0.011456,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "mistralai/mistral-medium-2505": {
                "accuracy": 82.902,
                "latency": 14.004,
                "stderr": 1.026,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "grok/grok-4-1-fast-non-reasoning": {
                "accuracy": 81.051,
                "latency": 3.881,
                "stderr": 1.066,
                "cost_per_test": 0.000279,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-3-5-sonnet-20241022": {
                "accuracy": 81.051,
                "latency": 6.366,
                "stderr": 1.066,
                "cost_per_test": 0.010386,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-2-1212": {
                "accuracy": 80.977,
                "latency": 12.63,
                "stderr": 1.068,
                "cost_per_test": 0.008117,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 0,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-2.0-flash-001": {
                "accuracy": 80.755,
                "latency": 8.408,
                "stderr": 1.073,
                "cost_per_test": 0.000825,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-1.5-flash-002": {
                "accuracy": 79.201,
                "latency": 2.126,
                "stderr": 1.113,
                "cost_per_test": 0.000223,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-4o-2024-08-06": {
                "accuracy": 78.905,
                "latency": 11.455,
                "stderr": 1.11,
                "cost_per_test": 0.009151,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "cohere/command-a-03-2025": {
                "accuracy": 78.46,
                "latency": 13.483,
                "stderr": 1.133,
                "cost_per_test": 0.009716,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 8000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "openai/gpt-4.1-nano-2025-04-14": {
                "accuracy": 78.238,
                "latency": 2.936,
                "stderr": 1.162,
                "cost_per_test": 0.000442,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "together/meta-llama/Llama-4-Scout-17B-16E-Instruct": {
                "accuracy": 75.944,
                "latency": 5.799,
                "stderr": 1.163,
                "cost_per_test": 0.0006,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "together/langston/nim/nvidia/llama-3.3-nemotron-super-49b-v1-42e84561": {
                "accuracy": 75.87,
                "latency": 15.477,
                "stderr": 1.189,
                "cost_per_test": null,
                "temperature": 0.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "mistralai/mistral-large-2411": {
                "accuracy": 75.722,
                "latency": 10.179,
                "stderr": 1.167,
                "cost_per_test": 0.006698,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "together/meta-llama/Llama-3.3-70B-Instruct-Turbo": {
                "accuracy": 75.5,
                "latency": 3.932,
                "stderr": 1.17,
                "cost_per_test": 0.001958,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "grok/grok-4.3": {
                "accuracy": 74.538,
                "latency": 332.566,
                "stderr": 0.739,
                "cost_per_test": 0.008528,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-4o-mini-2024-07-18": {
                "accuracy": 74.241,
                "latency": 6.517,
                "stderr": 1.19,
                "cost_per_test": 0.000567,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-4o-2024-11-20": {
                "accuracy": 73.871,
                "latency": 10.95,
                "stderr": 1.263,
                "cost_per_test": 0.009548,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "mistralai/mistral-small-2503": {
                "accuracy": 73.205,
                "latency": 4.976,
                "stderr": 1.205,
                "cost_per_test": 0.000274,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "mistralai/mistral-small-2402": {
                "accuracy": 72.835,
                "latency": 6.125,
                "stderr": 1.21,
                "cost_per_test": 0.000645,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "poolside/laguna-xs.2": {
                "accuracy": 68.246,
                "latency": 45.92,
                "stderr": 1.306,
                "cost_per_test": 0.000619,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "grok/grok-4-fast-non-reasoning": {
                "accuracy": 67.876,
                "latency": 1.575,
                "stderr": 1.27,
                "cost_per_test": 0.001346,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-3-5-haiku-20241022": {
                "accuracy": 66.173,
                "latency": 5.628,
                "stderr": 1.287,
                "cost_per_test": 0.002685,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "poolside/laguna-m.1": {
                "accuracy": 59.511,
                "latency": 119.023,
                "stderr": 1.339,
                "cost_per_test": 0.002116,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "ai21labs/jamba-large-1.6": {
                "accuracy": 58.105,
                "latency": 16.561,
                "stderr": 1.345,
                "cost_per_test": 0.006397,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "together/langston/nim/nvidia/llama-3.3-nemotron-super-49b-v1-42e84561-thinking": {
                "accuracy": 56.477,
                "latency": 44.146,
                "stderr": 1.349,
                "cost_per_test": null,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "mistralai/magistral-medium-2509": {
                "accuracy": 49.741,
                "latency": 33.17,
                "stderr": 1.36,
                "cost_per_test": 0.015989,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "mistralai/magistral-small-2509": {
                "accuracy": 39.97,
                "latency": 33.945,
                "stderr": 1.333,
                "cost_per_test": 0.00496,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "cohere/command-r-plus": {
                "accuracy": 39.008,
                "latency": 4.218,
                "stderr": 1.327,
                "cost_per_test": 0.006778,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 4000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "ai21labs/jamba-mini-1.6": {
                "accuracy": 30.792,
                "latency": 4.284,
                "stderr": 1.256,
                "cost_per_test": 0.000583,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            }
        },
        "other": {
            "anthropic/claude-fable-5-1": {
                "accuracy": 89.935,
                "latency": 22.847,
                "stderr": 0.99,
                "cost_per_test": 0.096818,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-fable-5": {
                "accuracy": 89.827,
                "latency": 14.291,
                "stderr": 0.994,
                "cost_per_test": 0.052645,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 88.636,
                "latency": 15.066,
                "stderr": 1.044,
                "cost_per_test": 0.016267,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-5": {
                "accuracy": 87.987,
                "latency": 7.209,
                "stderr": 1.07,
                "cost_per_test": 0.019263,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 87.987,
                "latency": 16.616,
                "stderr": 1.07,
                "cost_per_test": 0.024161,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-4-1-20250805": {
                "accuracy": 87.771,
                "latency": 14.998,
                "stderr": 1.078,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-4-1-20250805-thinking": {
                "accuracy": 87.771,
                "latency": 16.333,
                "stderr": 1.078,
                "cost_per_test": 0.068106,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.6-sol": {
                "accuracy": 87.662,
                "latency": 10.365,
                "stderr": 1.082,
                "cost_per_test": 0.023076,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3.7-flash": {
                "accuracy": 87.554,
                "latency": 3.375,
                "stderr": 1.086,
                "cost_per_test": 0.00772,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.8-flash": {
                "accuracy": 87.554,
                "latency": 7.512,
                "stderr": 1.086,
                "cost_per_test": 0.019777,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.5-flash": {
                "accuracy": 87.554,
                "latency": 37.383,
                "stderr": 1.118,
                "cost_per_test": 0.016173,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-4-20250514": {
                "accuracy": 87.338,
                "latency": 8.322,
                "stderr": 1.094,
                "cost_per_test": 0.033476,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-4.6": {
                "accuracy": 87.229,
                "latency": 42.931,
                "stderr": 1.098,
                "cost_per_test": 0.012286,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "grok/grok-4.5": {
                "accuracy": 87.229,
                "latency": 310.979,
                "stderr": 1.11,
                "cost_per_test": 0.010007,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 86.905,
                "latency": 17.063,
                "stderr": 1.635,
                "cost_per_test": 0.016102,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 86.905,
                "latency": 24.514,
                "stderr": 1.11,
                "cost_per_test": 0.012398,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-4-7": {
                "accuracy": 86.905,
                "latency": 35.837,
                "stderr": 1.162,
                "cost_per_test": 0.031505,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 86.797,
                "latency": 33.861,
                "stderr": 1.114,
                "cost_per_test": 0.023843,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3.6-flash": {
                "accuracy": 86.472,
                "latency": 5.755,
                "stderr": 1.125,
                "cost_per_test": 0.009441,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 86.472,
                "latency": 26.512,
                "stderr": 1.564,
                "cost_per_test": 0.051891,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-4-8": {
                "accuracy": 86.147,
                "latency": 13.617,
                "stderr": 1.136,
                "cost_per_test": 0.030976,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3.7-max": {
                "accuracy": 86.039,
                "latency": 19.902,
                "stderr": 1.14,
                "cost_per_test": 0.010569,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "minimax/MiniMax-M2.1": {
                "accuracy": 86.039,
                "latency": 26.158,
                "stderr": 1.14,
                "cost_per_test": 0.001852,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 85.823,
                "latency": 22.934,
                "stderr": 1.22,
                "cost_per_test": 0.020915,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 85.714,
                "latency": 34.123,
                "stderr": 1.151,
                "cost_per_test": 0.00613,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 85.498,
                "latency": 19.814,
                "stderr": 1.176,
                "cost_per_test": 0.099581,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-sonnet-5": {
                "accuracy": 85.39,
                "latency": 26.94,
                "stderr": 1.22,
                "cost_per_test": 0.043786,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "meta/muse_spark_1_1": {
                "accuracy": 85.281,
                "latency": 16.9,
                "stderr": 1.18,
                "cost_per_test": 0.008698,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "alibaba/qwen3.8-max": {
                "accuracy": 85.281,
                "latency": 45.914,
                "stderr": 1.166,
                "cost_per_test": 0.014219,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 84.957,
                "latency": 12.591,
                "stderr": 1.18,
                "cost_per_test": 0.010452,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 84.957,
                "latency": 22.307,
                "stderr": 1.176,
                "cost_per_test": 0.015416,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "meta/muse_spark_1_2": {
                "accuracy": 84.957,
                "latency": 35.056,
                "stderr": 1.183,
                "cost_per_test": 0.00837,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 84.957,
                "latency": 98.047,
                "stderr": 1.176,
                "cost_per_test": 0.011346,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101": {
                "accuracy": 84.632,
                "latency": 6.863,
                "stderr": 1.186,
                "cost_per_test": 0.035681,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "meta/muse_spark": {
                "accuracy": 84.632,
                "latency": 32.647,
                "stderr": 1.186,
                "cost_per_test": 0.001278,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "zai/glm-5.1": {
                "accuracy": 84.524,
                "latency": 36.629,
                "stderr": 1.19,
                "cost_per_test": 0.005805,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "deepseek/deepseek-v4-pro-0813": {
                "accuracy": 84.524,
                "latency": 45.323,
                "stderr": 1.203,
                "cost_per_test": 0.008369,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "zai/glm-5.3": {
                "accuracy": 84.524,
                "latency": 54.257,
                "stderr": 1.19,
                "cost_per_test": 0.014785,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "grok/grok-4-0709": {
                "accuracy": 84.524,
                "latency": 83.034,
                "stderr": 1.19,
                "cost_per_test": 0.031888,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/o3-2025-04-16": {
                "accuracy": 84.416,
                "latency": 9.539,
                "stderr": 1.193,
                "cost_per_test": 0.007168,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 84.307,
                "latency": 7.093,
                "stderr": 1.197,
                "cost_per_test": 0.008299,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "zai/glm-5-thinking": {
                "accuracy": 84.199,
                "latency": 53.559,
                "stderr": 1.2,
                "cost_per_test": 0.006713,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "thinkingmachines/inkling": {
                "accuracy": 84.199,
                "latency": 54.199,
                "stderr": 1.2,
                "cost_per_test": 0.027172,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "grok/grok-4.3": {
                "accuracy": 84.199,
                "latency": 336.051,
                "stderr": 1.2,
                "cost_per_test": 0.00688,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 84.199,
                "latency": 420.745,
                "stderr": 1.2,
                "cost_per_test": 0.013978,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "openai/gpt-5.6-terra": {
                "accuracy": 84.091,
                "latency": 5.556,
                "stderr": 1.203,
                "cost_per_test": 0.006067,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "kimi/kimi-k3": {
                "accuracy": 83.983,
                "latency": 30.907,
                "stderr": 1.207,
                "cost_per_test": 0.012905,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "zai/glm-5.2": {
                "accuracy": 83.658,
                "latency": 34.143,
                "stderr": 1.216,
                "cost_per_test": 0.008226,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514-thinking": {
                "accuracy": 83.55,
                "latency": 21.986,
                "stderr": 1.22,
                "cost_per_test": 0.021591,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 83.333,
                "latency": 24.362,
                "stderr": 1.226,
                "cost_per_test": 0.004649,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "deepseek/deepseek-v4-flash-0731": {
                "accuracy": 83.225,
                "latency": 12.88,
                "stderr": 1.229,
                "cost_per_test": 0.001856,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "anthropic/claude-3-7-sonnet-20250219-thinking": {
                "accuracy": 83.225,
                "latency": 19.935,
                "stderr": 1.229,
                "cost_per_test": 0.022639,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.5-flash-lite": {
                "accuracy": 83.117,
                "latency": 3.379,
                "stderr": 1.232,
                "cost_per_test": 0.002828,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 83.117,
                "latency": 21.251,
                "stderr": 1.401,
                "cost_per_test": 0.022672,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "nvidia/nemotron-3-ultra-550b-a55b": {
                "accuracy": 82.9,
                "latency": 15.648,
                "stderr": 1.239,
                "cost_per_test": null,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Nvidia",
                "harness": null
            },
            "openai/gpt-5.6-luna": {
                "accuracy": 82.684,
                "latency": 14.438,
                "stderr": 1.278,
                "cost_per_test": 0.002099,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/o1-2024-12-17": {
                "accuracy": 82.684,
                "latency": 16.101,
                "stderr": 1.245,
                "cost_per_test": 0.081239,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-2.5-pro-exp-03-25": {
                "accuracy": 82.576,
                "latency": 12.065,
                "stderr": 1.248,
                "cost_per_test": 0.005927,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "zai/glm-5.3-flash": {
                "accuracy": 82.468,
                "latency": 23.566,
                "stderr": 1.251,
                "cost_per_test": 0.000529,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "fireworks/deepseek-v3p2-thinking": {
                "accuracy": 82.468,
                "latency": 59.92,
                "stderr": 1.251,
                "cost_per_test": 0.000662,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "alibaba/qwen3.5-plus-thinking": {
                "accuracy": 82.468,
                "latency": 62.772,
                "stderr": 1.251,
                "cost_per_test": 0.0061,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 82.359,
                "latency": 23.394,
                "stderr": 1.278,
                "cost_per_test": 0.024895,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3-max-2026-01-23": {
                "accuracy": 82.035,
                "latency": 80.23,
                "stderr": 1.263,
                "cost_per_test": 0.017908,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 81.926,
                "latency": 5.446,
                "stderr": 1.266,
                "cost_per_test": 0.000674,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514": {
                "accuracy": 81.71,
                "latency": 6.537,
                "stderr": 1.272,
                "cost_per_test": 0.007223,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "fireworks/deepseek-v3p2": {
                "accuracy": 81.602,
                "latency": 38.008,
                "stderr": 1.357,
                "cost_per_test": 0.000157,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "fireworks/deepseek-r1": {
                "accuracy": 81.385,
                "latency": 18.175,
                "stderr": 1.494,
                "cost_per_test": 0.008235,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "minimax/MiniMax-M3": {
                "accuracy": 81.277,
                "latency": 18.963,
                "stderr": 1.289,
                "cost_per_test": 0.002008,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 512000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "thinkingmachines/inkling-small": {
                "accuracy": 81.277,
                "latency": 45.035,
                "stderr": 1.283,
                "cost_per_test": 0.003006,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 262000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "anthropic/claude-3-7-sonnet-20250219": {
                "accuracy": 81.169,
                "latency": 5.192,
                "stderr": 1.286,
                "cost_per_test": 0.006876,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 81.061,
                "latency": 15.741,
                "stderr": 1.289,
                "cost_per_test": 0.000682,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "alibaba/qwen3-max-preview": {
                "accuracy": 80.844,
                "latency": 28.347,
                "stderr": 1.295,
                "cost_per_test": 0.003492,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "xiaomi/mimo-v2.5-pro": {
                "accuracy": 80.628,
                "latency": 36.392,
                "stderr": 1.582,
                "cost_per_test": 0.000878,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "alibaba/qwen3-max": {
                "accuracy": 80.519,
                "latency": 0.0,
                "stderr": 1.303,
                "cost_per_test": 0.003535,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 80.411,
                "latency": 16.063,
                "stderr": 1.322,
                "cost_per_test": 0.002935,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "kimi/kimi-k2-thinking": {
                "accuracy": 80.411,
                "latency": 67.872,
                "stderr": 1.33,
                "cost_per_test": 0.001378,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "zai/glm-4.5": {
                "accuracy": 80.411,
                "latency": 69.032,
                "stderr": 1.306,
                "cost_per_test": 0.005996,
                "temperature": 0.6,
                "top_p": 1.0,
                "max_output_tokens": 81920,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025": {
                "accuracy": 80.303,
                "latency": 4.209,
                "stderr": 1.308,
                "cost_per_test": 0.000805,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "zai/glm-4.7": {
                "accuracy": 80.303,
                "latency": 72.709,
                "stderr": 1.308,
                "cost_per_test": 0.004031,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "xiaomi/mimo-v2.5": {
                "accuracy": 79.978,
                "latency": 18.197,
                "stderr": 1.324,
                "cost_per_test": 0.000405,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "openai/gpt-4.1-2025-04-14": {
                "accuracy": 79.87,
                "latency": 2.19,
                "stderr": 1.319,
                "cost_per_test": 0.00323,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4-fast-reasoning": {
                "accuracy": 79.654,
                "latency": 4.164,
                "stderr": 1.324,
                "cost_per_test": 0.001093,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-3-5-sonnet-20241022": {
                "accuracy": 79.545,
                "latency": 5.76,
                "stderr": 1.327,
                "cost_per_test": 0.006445,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025-thinking": {
                "accuracy": 79.545,
                "latency": 7.515,
                "stderr": 1.327,
                "cost_per_test": 0.000798,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-3-mini-fast-high-reasoning": {
                "accuracy": 79.545,
                "latency": 7.928,
                "stderr": 1.327,
                "cost_per_test": 0.000975,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 79.545,
                "latency": 14.719,
                "stderr": 1.327,
                "cost_per_test": 0.003481,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-4.6": {
                "accuracy": 79.545,
                "latency": 34.003,
                "stderr": 1.327,
                "cost_per_test": 0.004223,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 122880,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "grok/grok-3": {
                "accuracy": 79.221,
                "latency": 8.568,
                "stderr": 1.335,
                "cost_per_test": 0.011647,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "mistralai/mistral-large-2512": {
                "accuracy": 79.221,
                "latency": 8.956,
                "stderr": 1.632,
                "cost_per_test": 0.000929,
                "temperature": 0.065,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "alibaba/qwen3.5-flash": {
                "accuracy": 78.788,
                "latency": 25.541,
                "stderr": 1.345,
                "cost_per_test": 0.001274,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "minimax/MiniMax-M2.5": {
                "accuracy": 78.788,
                "latency": 30.649,
                "stderr": 1.345,
                "cost_per_test": 0.003394,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "grok/grok-3-mini-fast-low-reasoning": {
                "accuracy": 78.68,
                "latency": 4.975,
                "stderr": 1.347,
                "cost_per_test": 0.000712,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "low",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/o4-mini-2025-04-16": {
                "accuracy": 78.68,
                "latency": 7.297,
                "stderr": 1.347,
                "cost_per_test": 0.004122,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "fireworks/deepseek-v3-0324": {
                "accuracy": 78.247,
                "latency": 9.885,
                "stderr": 1.357,
                "cost_per_test": 0.000955,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "mistralai/mistral-medium-3.5": {
                "accuracy": 78.139,
                "latency": 39.948,
                "stderr": 1.36,
                "cost_per_test": 0.026514,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "fireworks/qwen3-235b-a22b": {
                "accuracy": 78.03,
                "latency": 40.542,
                "stderr": 1.364,
                "cost_per_test": 0.001379,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "together/moonshotai/Kimi-K2-Instruct": {
                "accuracy": 77.922,
                "latency": 10.042,
                "stderr": 1.364,
                "cost_per_test": 0.001608,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "fireworks/llama4-maverick-instruct-basic": {
                "accuracy": 77.706,
                "latency": 5.163,
                "stderr": 1.369,
                "cost_per_test": 0.000541,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 77.489,
                "latency": 36.909,
                "stderr": 1.386,
                "cost_per_test": 0.002513,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "openai/gpt-4o-2024-11-20": {
                "accuracy": 77.273,
                "latency": 5.526,
                "stderr": 1.397,
                "cost_per_test": 0.004344,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-4o-2024-08-06": {
                "accuracy": 77.165,
                "latency": 5.932,
                "stderr": 1.381,
                "cost_per_test": 0.004937,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025-thinking": {
                "accuracy": 76.84,
                "latency": 8.282,
                "stderr": 1.388,
                "cost_per_test": 0.00019,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3.8-27b": {
                "accuracy": 76.84,
                "latency": 20.061,
                "stderr": 1.388,
                "cost_per_test": 0.007396,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "grok/grok-2-1212": {
                "accuracy": 76.623,
                "latency": 3.619,
                "stderr": 1.392,
                "cost_per_test": 0.003518,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 0,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025": {
                "accuracy": 76.407,
                "latency": 2.052,
                "stderr": 1.397,
                "cost_per_test": 0.0003,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "ant/ling-3.0-flash-2607": {
                "accuracy": 76.19,
                "latency": 9.149,
                "stderr": 1.401,
                "cost_per_test": 0.000353,
                "temperature": 0.6,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "mistralai/mistral-medium-2505": {
                "accuracy": 76.082,
                "latency": 3.419,
                "stderr": 1.41,
                "cost_per_test": 0.00077,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "mistralai/magistral-medium-2509": {
                "accuracy": 76.082,
                "latency": 29.004,
                "stderr": 1.403,
                "cost_per_test": 0.009138,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "openai/gpt-4.1-mini-2025-04-14": {
                "accuracy": 75.758,
                "latency": 2.07,
                "stderr": 1.41,
                "cost_per_test": 0.000755,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 75.758,
                "latency": 14.09,
                "stderr": 1.64,
                "cost_per_test": 0.008857,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-2.0-flash-001": {
                "accuracy": 75.325,
                "latency": 1.702,
                "stderr": 1.418,
                "cost_per_test": 0.000172,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-1.5-pro-002": {
                "accuracy": 75.0,
                "latency": 1.857,
                "stderr": 1.425,
                "cost_per_test": 0.0016,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-4-1-fast-non-reasoning": {
                "accuracy": 74.351,
                "latency": 1.454,
                "stderr": 1.437,
                "cost_per_test": 0.000127,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "poolside/laguna-m.1": {
                "accuracy": 74.351,
                "latency": 70.735,
                "stderr": 1.463,
                "cost_per_test": 0.000805,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "openai/o3-mini-2025-01-31": {
                "accuracy": 74.242,
                "latency": 17.041,
                "stderr": 1.439,
                "cost_per_test": 0.010356,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "fireworks/deepseek-v3": {
                "accuracy": 74.134,
                "latency": 4.106,
                "stderr": 1.441,
                "cost_per_test": 0.000891,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "fireworks/gpt-oss-120b": {
                "accuracy": 73.81,
                "latency": 38.694,
                "stderr": 1.448,
                "cost_per_test": 0.001439,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "together/langston/nim/nvidia/llama-3.3-nemotron-super-49b-v1-42e84561-thinking": {
                "accuracy": 73.701,
                "latency": 30.305,
                "stderr": 1.454,
                "cost_per_test": null,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "grok/grok-4-fast-non-reasoning": {
                "accuracy": 73.052,
                "latency": 4.498,
                "stderr": 1.46,
                "cost_per_test": 0.000764,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-5-nano-2025-08-07": {
                "accuracy": 72.619,
                "latency": 30.014,
                "stderr": 1.467,
                "cost_per_test": 0.00129,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 71.861,
                "latency": 5.742,
                "stderr": 1.557,
                "cost_per_test": 0.000192,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "together/meta-llama/Llama-3.3-70B-Instruct-Turbo": {
                "accuracy": 71.429,
                "latency": 3.776,
                "stderr": 1.486,
                "cost_per_test": 0.001113,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "mistralai/mistral-large-2411": {
                "accuracy": 70.671,
                "latency": 3.589,
                "stderr": 1.498,
                "cost_per_test": 0.003046,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "cohere/command-a-03-2025": {
                "accuracy": 69.481,
                "latency": 6.18,
                "stderr": 1.563,
                "cost_per_test": 0.005133,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 8000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "poolside/laguna-xs.2": {
                "accuracy": 67.424,
                "latency": 26.896,
                "stderr": 1.617,
                "cost_per_test": 0.000284,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "anthropic/claude-3-5-haiku-20241022": {
                "accuracy": 66.991,
                "latency": 5.582,
                "stderr": 1.547,
                "cost_per_test": 0.001749,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "together/langston/nim/nvidia/llama-3.3-nemotron-super-49b-v1-42e84561": {
                "accuracy": 66.775,
                "latency": 12.585,
                "stderr": 1.619,
                "cost_per_test": null,
                "temperature": 0.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "mistralai/mistral-small-2503": {
                "accuracy": 66.45,
                "latency": 1.757,
                "stderr": 1.553,
                "cost_per_test": 0.00012,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "openai/gpt-4o-mini-2024-07-18": {
                "accuracy": 66.45,
                "latency": 2.835,
                "stderr": 1.553,
                "cost_per_test": 0.000253,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "together/meta-llama/Llama-4-Scout-17B-16E-Instruct": {
                "accuracy": 66.126,
                "latency": 3.433,
                "stderr": 1.557,
                "cost_per_test": 0.000353,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "fireworks/gpt-oss-20b": {
                "accuracy": 66.126,
                "latency": 16.455,
                "stderr": 1.576,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "mistralai/mistral-small-2402": {
                "accuracy": 65.043,
                "latency": 2.356,
                "stderr": 1.569,
                "cost_per_test": 0.0003,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "openai/gpt-4.1-nano-2025-04-14": {
                "accuracy": 64.394,
                "latency": 1.481,
                "stderr": 1.577,
                "cost_per_test": 0.000196,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-1.5-flash-002": {
                "accuracy": 62.338,
                "latency": 1.045,
                "stderr": 1.594,
                "cost_per_test": 0.000108,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "mistralai/magistral-small-2509": {
                "accuracy": 54.545,
                "latency": 9.55,
                "stderr": 1.638,
                "cost_per_test": 0.002112,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "cohere/command-r-plus": {
                "accuracy": 51.082,
                "latency": 1.729,
                "stderr": 1.645,
                "cost_per_test": 0.003163,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 4000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "ai21labs/jamba-large-1.6": {
                "accuracy": 44.048,
                "latency": 4.044,
                "stderr": 1.629,
                "cost_per_test": 0.002706,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "ai21labs/jamba-mini-1.6": {
                "accuracy": 41.991,
                "latency": 1.232,
                "stderr": 1.624,
                "cost_per_test": 0.00023,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            }
        },
        "philosophy": {
            "anthropic/claude-fable-5": {
                "accuracy": 90.581,
                "latency": 24.035,
                "stderr": 1.308,
                "cost_per_test": 0.069071,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 90.381,
                "latency": 15.572,
                "stderr": 1.32,
                "cost_per_test": 0.017428,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-fable-5-1": {
                "accuracy": 90.18,
                "latency": 27.85,
                "stderr": 1.332,
                "cost_per_test": 0.117235,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.7-flash": {
                "accuracy": 89.379,
                "latency": 3.821,
                "stderr": 1.379,
                "cost_per_test": 0.00827,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 89.379,
                "latency": 24.329,
                "stderr": 1.391,
                "cost_per_test": 0.124834,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3.7-max": {
                "accuracy": 89.178,
                "latency": 21.703,
                "stderr": 1.391,
                "cost_per_test": 0.01142,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "grok/grok-4.6": {
                "accuracy": 89.178,
                "latency": 43.038,
                "stderr": 1.391,
                "cost_per_test": 0.011442,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-opus-5": {
                "accuracy": 88.978,
                "latency": 8.992,
                "stderr": 1.402,
                "cost_per_test": 0.022728,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 88.978,
                "latency": 16.751,
                "stderr": 1.402,
                "cost_per_test": 0.024276,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-4-7": {
                "accuracy": 88.978,
                "latency": 195.014,
                "stderr": 1.456,
                "cost_per_test": 0.030079,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 88.377,
                "latency": 19.566,
                "stderr": 2.136,
                "cost_per_test": 0.01872,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.6-sol": {
                "accuracy": 88.377,
                "latency": 24.805,
                "stderr": 1.435,
                "cost_per_test": 0.025924,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3.8-max": {
                "accuracy": 88.377,
                "latency": 44.233,
                "stderr": 1.435,
                "cost_per_test": 0.013668,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "google/gemini-3.8-flash": {
                "accuracy": 88.176,
                "latency": 7.866,
                "stderr": 1.445,
                "cost_per_test": 0.019743,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-4-1-20250805-thinking": {
                "accuracy": 88.176,
                "latency": 20.153,
                "stderr": 1.445,
                "cost_per_test": 0.084337,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101": {
                "accuracy": 87.776,
                "latency": 7.596,
                "stderr": 1.477,
                "cost_per_test": 0.040147,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 87.776,
                "latency": 34.673,
                "stderr": 1.466,
                "cost_per_test": 0.025903,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3.6-flash": {
                "accuracy": 87.575,
                "latency": 6.594,
                "stderr": 1.477,
                "cost_per_test": 0.010596,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.5-flash": {
                "accuracy": 87.575,
                "latency": 10.617,
                "stderr": 1.477,
                "cost_per_test": 0.01891,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-4-1-20250805": {
                "accuracy": 87.575,
                "latency": 21.155,
                "stderr": 1.477,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-4.5": {
                "accuracy": 87.575,
                "latency": 356.138,
                "stderr": 1.477,
                "cost_per_test": 0.009439,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "meta/muse_spark_1_2": {
                "accuracy": 87.375,
                "latency": 34.78,
                "stderr": 1.487,
                "cost_per_test": 0.008642,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "meta/muse_spark_1_1": {
                "accuracy": 87.174,
                "latency": 18.405,
                "stderr": 1.507,
                "cost_per_test": 0.009215,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 87.174,
                "latency": 38.14,
                "stderr": 1.497,
                "cost_per_test": 0.006638,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 86.974,
                "latency": 50.281,
                "stderr": 1.908,
                "cost_per_test": 0.047208,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4.3": {
                "accuracy": 86.774,
                "latency": 1170.986,
                "stderr": 1.517,
                "cost_per_test": 0.006207,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-opus-4-8": {
                "accuracy": 86.573,
                "latency": 12.119,
                "stderr": 1.526,
                "cost_per_test": 0.031561,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-sonnet-5": {
                "accuracy": 86.573,
                "latency": 25.082,
                "stderr": 1.536,
                "cost_per_test": 0.039802,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 86.172,
                "latency": 20.743,
                "stderr": 1.97,
                "cost_per_test": 0.02245,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 86.172,
                "latency": 25.258,
                "stderr": 1.545,
                "cost_per_test": 0.016946,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 85.972,
                "latency": 18.756,
                "stderr": 1.555,
                "cost_per_test": 0.011306,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "minimax/MiniMax-M2.1": {
                "accuracy": 85.972,
                "latency": 23.972,
                "stderr": 1.555,
                "cost_per_test": 0.001774,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 85.972,
                "latency": 26.471,
                "stderr": 1.582,
                "cost_per_test": 0.025073,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-4-20250514": {
                "accuracy": 85.571,
                "latency": 9.721,
                "stderr": 1.573,
                "cost_per_test": 0.039418,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 85.571,
                "latency": 24.271,
                "stderr": 1.609,
                "cost_per_test": 0.024711,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3.5-plus-thinking": {
                "accuracy": 85.571,
                "latency": 60.923,
                "stderr": 1.573,
                "cost_per_test": 0.00662,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 85.371,
                "latency": 27.179,
                "stderr": 1.582,
                "cost_per_test": 0.013969,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "meta/muse_spark": {
                "accuracy": 85.371,
                "latency": 34.617,
                "stderr": 1.582,
                "cost_per_test": 0.001328,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "kimi/kimi-k3": {
                "accuracy": 85.371,
                "latency": 41.629,
                "stderr": 1.582,
                "cost_per_test": 0.016284,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "openai/o3-2025-04-16": {
                "accuracy": 85.17,
                "latency": 14.017,
                "stderr": 1.591,
                "cost_per_test": 0.007963,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "deepseek/deepseek-v4-pro-0813": {
                "accuracy": 84.569,
                "latency": 35.176,
                "stderr": 1.729,
                "cost_per_test": 0.00849,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "zai/glm-5.3-flash": {
                "accuracy": 84.569,
                "latency": 36.864,
                "stderr": 1.617,
                "cost_per_test": 0.000892,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 84.569,
                "latency": 70.384,
                "stderr": 1.617,
                "cost_per_test": 0.011206,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "zai/glm-5.3": {
                "accuracy": 84.569,
                "latency": 82.002,
                "stderr": 1.617,
                "cost_per_test": 0.020607,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 84.569,
                "latency": 657.825,
                "stderr": 1.617,
                "cost_per_test": 0.014343,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "zai/glm-5.2": {
                "accuracy": 84.168,
                "latency": 43.916,
                "stderr": 1.634,
                "cost_per_test": 0.011096,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "openai/gpt-5.6-terra": {
                "accuracy": 83.968,
                "latency": 6.126,
                "stderr": 1.642,
                "cost_per_test": 0.006548,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-5.1": {
                "accuracy": 83.936,
                "latency": 32.773,
                "stderr": 1.645,
                "cost_per_test": 0.006152,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 83.367,
                "latency": 6.419,
                "stderr": 1.667,
                "cost_per_test": 0.008539,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "deepseek/deepseek-v4-flash-0731": {
                "accuracy": 83.367,
                "latency": 16.678,
                "stderr": 1.667,
                "cost_per_test": 0.002161,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "grok/grok-4-0709": {
                "accuracy": 83.367,
                "latency": 57.631,
                "stderr": 1.667,
                "cost_per_test": 0.029648,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "zai/glm-5-thinking": {
                "accuracy": 83.166,
                "latency": 73.251,
                "stderr": 1.683,
                "cost_per_test": 0.00732,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "google/gemini-3.5-flash-lite": {
                "accuracy": 82.966,
                "latency": 4.243,
                "stderr": 1.683,
                "cost_per_test": 0.003441,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.6-luna": {
                "accuracy": 82.766,
                "latency": 15.843,
                "stderr": 1.698,
                "cost_per_test": 0.002589,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-2.5-pro-exp-03-25": {
                "accuracy": 82.565,
                "latency": 15.171,
                "stderr": 1.698,
                "cost_per_test": 0.00737,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 82.164,
                "latency": 26.244,
                "stderr": 1.714,
                "cost_per_test": 0.005272,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "thinkingmachines/inkling": {
                "accuracy": 82.164,
                "latency": 51.445,
                "stderr": 1.714,
                "cost_per_test": 0.025216,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 81.764,
                "latency": 6.827,
                "stderr": 1.729,
                "cost_per_test": 0.00079,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514-thinking": {
                "accuracy": 81.563,
                "latency": 38.216,
                "stderr": 1.736,
                "cost_per_test": 0.033281,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3-max-2026-01-23": {
                "accuracy": 81.563,
                "latency": 79.18,
                "stderr": 1.736,
                "cost_per_test": 0.017179,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 81.162,
                "latency": 14.651,
                "stderr": 1.75,
                "cost_per_test": 0.00065,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "nvidia/nemotron-3-ultra-550b-a55b": {
                "accuracy": 80.962,
                "latency": 16.114,
                "stderr": 1.758,
                "cost_per_test": null,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Nvidia",
                "harness": null
            },
            "openai/o1-2024-12-17": {
                "accuracy": 80.962,
                "latency": 19.121,
                "stderr": 1.758,
                "cost_per_test": 0.086531,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514": {
                "accuracy": 80.762,
                "latency": 10.531,
                "stderr": 1.765,
                "cost_per_test": 0.008185,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 80.561,
                "latency": 15.236,
                "stderr": 1.785,
                "cost_per_test": 0.003351,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "thinkingmachines/inkling-small": {
                "accuracy": 80.561,
                "latency": 52.89,
                "stderr": 1.772,
                "cost_per_test": 0.003557,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 262000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "fireworks/deepseek-v3p2-thinking": {
                "accuracy": 80.561,
                "latency": 61.44,
                "stderr": 1.772,
                "cost_per_test": 0.000613,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "minimax/MiniMax-M3": {
                "accuracy": 80.16,
                "latency": 23.038,
                "stderr": 1.792,
                "cost_per_test": 0.002277,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 512000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "zai/glm-4.5": {
                "accuracy": 80.16,
                "latency": 74.109,
                "stderr": 1.785,
                "cost_per_test": 0.005902,
                "temperature": 0.6,
                "top_p": 1.0,
                "max_output_tokens": 81920,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "xiaomi/mimo-v2.5-pro": {
                "accuracy": 79.96,
                "latency": 32.952,
                "stderr": 2.116,
                "cost_per_test": 0.000991,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "anthropic/claude-3-7-sonnet-20250219-thinking": {
                "accuracy": 79.76,
                "latency": 24.985,
                "stderr": 1.799,
                "cost_per_test": 0.029247,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "kimi/kimi-k2-thinking": {
                "accuracy": 79.559,
                "latency": 85.402,
                "stderr": 1.812,
                "cost_per_test": 0.001517,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025": {
                "accuracy": 79.359,
                "latency": 5.223,
                "stderr": 1.812,
                "cost_per_test": 0.000995,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "fireworks/deepseek-v3p2": {
                "accuracy": 79.158,
                "latency": 51.886,
                "stderr": 1.831,
                "cost_per_test": 0.000163,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "grok/grok-4-fast-reasoning": {
                "accuracy": 78.958,
                "latency": 3.994,
                "stderr": 1.825,
                "cost_per_test": 0.001108,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "alibaba/qwen3.5-flash": {
                "accuracy": 78.958,
                "latency": 36.885,
                "stderr": 1.825,
                "cost_per_test": 0.00145,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025-thinking": {
                "accuracy": 78.758,
                "latency": 8.332,
                "stderr": 1.831,
                "cost_per_test": 0.001001,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "xiaomi/mimo-v2.5": {
                "accuracy": 78.758,
                "latency": 20.433,
                "stderr": 1.837,
                "cost_per_test": 0.000429,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "alibaba/qwen3-max": {
                "accuracy": 78.557,
                "latency": 0.0,
                "stderr": 1.837,
                "cost_per_test": 0.003992,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "alibaba/qwen3.8-27b": {
                "accuracy": 78.557,
                "latency": 26.721,
                "stderr": 1.837,
                "cost_per_test": 0.007942,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "alibaba/qwen3-max-preview": {
                "accuracy": 78.557,
                "latency": 34.792,
                "stderr": 1.837,
                "cost_per_test": 0.004204,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "zai/glm-4.7": {
                "accuracy": 77.956,
                "latency": 77.484,
                "stderr": 1.862,
                "cost_per_test": 0.005365,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "anthropic/claude-3-7-sonnet-20250219": {
                "accuracy": 77.154,
                "latency": 6.058,
                "stderr": 1.879,
                "cost_per_test": 0.007848,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "fireworks/deepseek-r1": {
                "accuracy": 76.553,
                "latency": 17.808,
                "stderr": 2.072,
                "cost_per_test": 0.008096,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "mistralai/mistral-large-2512": {
                "accuracy": 76.353,
                "latency": 9.977,
                "stderr": 2.169,
                "cost_per_test": 0.001026,
                "temperature": 0.065,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "openai/gpt-4.1-2025-04-14": {
                "accuracy": 76.353,
                "latency": 18.234,
                "stderr": 1.902,
                "cost_per_test": 0.013185,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-4.6": {
                "accuracy": 76.353,
                "latency": 35.981,
                "stderr": 1.902,
                "cost_per_test": 0.005123,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 122880,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "anthropic/claude-3-5-sonnet-20241022": {
                "accuracy": 76.152,
                "latency": 6.094,
                "stderr": 1.908,
                "cost_per_test": 0.007086,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "fireworks/deepseek-v3-0324": {
                "accuracy": 75.752,
                "latency": 24.442,
                "stderr": 1.919,
                "cost_per_test": 0.000964,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "ant/ling-3.0-flash-2607": {
                "accuracy": 75.551,
                "latency": 4.735,
                "stderr": 1.924,
                "cost_per_test": 0.000208,
                "temperature": 0.6,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "grok/grok-3-mini-fast-low-reasoning": {
                "accuracy": 75.551,
                "latency": 5.384,
                "stderr": 1.924,
                "cost_per_test": 0.000764,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "low",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 75.351,
                "latency": 15.556,
                "stderr": 1.929,
                "cost_per_test": 0.00387,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-3-mini-fast-high-reasoning": {
                "accuracy": 74.95,
                "latency": 8.91,
                "stderr": 1.94,
                "cost_per_test": 0.001047,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 74.549,
                "latency": 18.668,
                "stderr": 2.231,
                "cost_per_test": 0.012033,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-3": {
                "accuracy": 74.148,
                "latency": 13.985,
                "stderr": 1.96,
                "cost_per_test": 0.013679,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "fireworks/llama4-maverick-instruct-basic": {
                "accuracy": 73.747,
                "latency": 4.691,
                "stderr": 1.97,
                "cost_per_test": 0.000595,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "together/moonshotai/Kimi-K2-Instruct": {
                "accuracy": 73.747,
                "latency": 9.06,
                "stderr": 1.97,
                "cost_per_test": 0.001606,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 73.747,
                "latency": 40.819,
                "stderr": 1.989,
                "cost_per_test": 0.002607,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "openai/o4-mini-2025-04-16": {
                "accuracy": 73.547,
                "latency": 11.216,
                "stderr": 1.975,
                "cost_per_test": 0.004966,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025": {
                "accuracy": 73.347,
                "latency": 2.348,
                "stderr": 1.979,
                "cost_per_test": 0.000367,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "mistralai/mistral-medium-3.5": {
                "accuracy": 73.146,
                "latency": 52.443,
                "stderr": 1.984,
                "cost_per_test": 0.029496,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "grok/grok-4-1-fast-non-reasoning": {
                "accuracy": 72.946,
                "latency": 2.103,
                "stderr": 1.989,
                "cost_per_test": 0.000151,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "fireworks/qwen3-235b-a22b": {
                "accuracy": 72.345,
                "latency": 40.377,
                "stderr": 2.028,
                "cost_per_test": 0.001476,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025-thinking": {
                "accuracy": 71.944,
                "latency": 7.43,
                "stderr": 2.011,
                "cost_per_test": 0.000237,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-2.0-flash-001": {
                "accuracy": 71.543,
                "latency": 2.256,
                "stderr": 2.02,
                "cost_per_test": 0.00022,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "mistralai/mistral-medium-2505": {
                "accuracy": 71.543,
                "latency": 5.458,
                "stderr": 2.028,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "openai/gpt-4o-2024-11-20": {
                "accuracy": 71.142,
                "latency": 6.384,
                "stderr": 2.056,
                "cost_per_test": 0.005199,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-1.5-pro-002": {
                "accuracy": 70.341,
                "latency": 2.096,
                "stderr": 2.045,
                "cost_per_test": 0.001742,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-2-1212": {
                "accuracy": 70.341,
                "latency": 4.414,
                "stderr": 2.045,
                "cost_per_test": 0.003941,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 0,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "fireworks/deepseek-v3": {
                "accuracy": 70.341,
                "latency": 11.205,
                "stderr": 2.045,
                "cost_per_test": 0.000936,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "mistralai/magistral-medium-2509": {
                "accuracy": 70.14,
                "latency": 22.523,
                "stderr": 2.049,
                "cost_per_test": 0.009835,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "grok/grok-4-fast-non-reasoning": {
                "accuracy": 69.756,
                "latency": 2.738,
                "stderr": 1.435,
                "cost_per_test": 0.000827,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "minimax/MiniMax-M2.5": {
                "accuracy": 69.539,
                "latency": 38.873,
                "stderr": 2.064,
                "cost_per_test": 0.00379,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "openai/o3-mini-2025-01-31": {
                "accuracy": 69.138,
                "latency": 20.18,
                "stderr": 2.068,
                "cost_per_test": 0.009987,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-4.1-mini-2025-04-14": {
                "accuracy": 68.737,
                "latency": 2.836,
                "stderr": 2.075,
                "cost_per_test": 0.00094,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-4o-2024-08-06": {
                "accuracy": 67.535,
                "latency": 5.941,
                "stderr": 2.096,
                "cost_per_test": 0.005383,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "fireworks/gpt-oss-120b": {
                "accuracy": 67.134,
                "latency": 10.104,
                "stderr": 2.103,
                "cost_per_test": 0.00128,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "together/langston/nim/nvidia/llama-3.3-nemotron-super-49b-v1-42e84561-thinking": {
                "accuracy": 66.533,
                "latency": 26.197,
                "stderr": 2.116,
                "cost_per_test": null,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "poolside/laguna-m.1": {
                "accuracy": 66.333,
                "latency": 32.43,
                "stderr": 2.133,
                "cost_per_test": 0.000828,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 66.132,
                "latency": 5.847,
                "stderr": 2.172,
                "cost_per_test": 0.00023,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5-nano-2025-08-07": {
                "accuracy": 65.13,
                "latency": 22.338,
                "stderr": 2.133,
                "cost_per_test": 0.001592,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "mistralai/mistral-large-2411": {
                "accuracy": 63.727,
                "latency": 4.929,
                "stderr": 2.152,
                "cost_per_test": 0.003555,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "cohere/command-a-03-2025": {
                "accuracy": 62.124,
                "latency": 7.621,
                "stderr": 2.178,
                "cost_per_test": 0.005916,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 8000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "together/langston/nim/nvidia/llama-3.3-nemotron-super-49b-v1-42e84561": {
                "accuracy": 61.723,
                "latency": 15.884,
                "stderr": 2.21,
                "cost_per_test": null,
                "temperature": 0.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "together/meta-llama/Llama-3.3-70B-Instruct-Turbo": {
                "accuracy": 61.523,
                "latency": 4.873,
                "stderr": 2.178,
                "cost_per_test": 0.001224,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "poolside/laguna-xs.2": {
                "accuracy": 58.918,
                "latency": 31.595,
                "stderr": 2.237,
                "cost_per_test": 0.000303,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "fireworks/gpt-oss-20b": {
                "accuracy": 58.717,
                "latency": 24.252,
                "stderr": 2.214,
                "cost_per_test": 0.000851,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "mistralai/mistral-small-2402": {
                "accuracy": 58.517,
                "latency": 3.092,
                "stderr": 2.206,
                "cost_per_test": 0.000351,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "anthropic/claude-3-5-haiku-20241022": {
                "accuracy": 58.317,
                "latency": 5.723,
                "stderr": 2.207,
                "cost_per_test": 0.001843,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "mistralai/magistral-small-2509": {
                "accuracy": 57.515,
                "latency": 10.587,
                "stderr": 2.213,
                "cost_per_test": 0.002458,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "together/meta-llama/Llama-4-Scout-17B-16E-Instruct": {
                "accuracy": 57.315,
                "latency": 4.895,
                "stderr": 2.214,
                "cost_per_test": 0.00041,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "google/gemini-1.5-flash-002": {
                "accuracy": 56.513,
                "latency": 1.282,
                "stderr": 2.219,
                "cost_per_test": 0.000123,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "mistralai/mistral-small-2503": {
                "accuracy": 55.311,
                "latency": 2.479,
                "stderr": 2.227,
                "cost_per_test": 0.000136,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "openai/gpt-4.1-nano-2025-04-14": {
                "accuracy": 52.906,
                "latency": 1.806,
                "stderr": 2.237,
                "cost_per_test": 0.000241,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-4o-mini-2024-07-18": {
                "accuracy": 50.701,
                "latency": 4.171,
                "stderr": 2.238,
                "cost_per_test": 0.000319,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "ai21labs/jamba-large-1.6": {
                "accuracy": 47.295,
                "latency": 4.55,
                "stderr": 2.235,
                "cost_per_test": 0.003064,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "cohere/command-r-plus": {
                "accuracy": 42.084,
                "latency": 2.517,
                "stderr": 2.209,
                "cost_per_test": 0.003737,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 4000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "ai21labs/jamba-mini-1.6": {
                "accuracy": 35.671,
                "latency": 1.912,
                "stderr": 2.144,
                "cost_per_test": 0.000301,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            }
        },
        "physics": {
            "anthropic/claude-fable-5-1": {
                "accuracy": 96.69,
                "latency": 17.878,
                "stderr": 0.496,
                "cost_per_test": 0.086595,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-fable-5": {
                "accuracy": 95.843,
                "latency": 15.9,
                "stderr": 0.554,
                "cost_per_test": 0.071684,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.8-flash": {
                "accuracy": 94.765,
                "latency": 7.952,
                "stderr": 0.618,
                "cost_per_test": 0.024694,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-5": {
                "accuracy": 94.765,
                "latency": 8.999,
                "stderr": 0.618,
                "cost_per_test": 0.025532,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 94.688,
                "latency": 35.667,
                "stderr": 0.622,
                "cost_per_test": 0.038713,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.7-flash": {
                "accuracy": 94.303,
                "latency": 4.242,
                "stderr": 0.643,
                "cost_per_test": 0.012698,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 94.149,
                "latency": 61.982,
                "stderr": 1.175,
                "cost_per_test": 0.066938,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.5-flash": {
                "accuracy": 94.072,
                "latency": 58.125,
                "stderr": 0.663,
                "cost_per_test": 0.029412,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "meta/muse_spark_1_1": {
                "accuracy": 93.841,
                "latency": 21.651,
                "stderr": 0.69,
                "cost_per_test": 0.011016,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 93.687,
                "latency": 29.722,
                "stderr": 0.675,
                "cost_per_test": 0.046396,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-4-8": {
                "accuracy": 93.533,
                "latency": 23.621,
                "stderr": 0.682,
                "cost_per_test": 0.061096,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3.7-max": {
                "accuracy": 93.533,
                "latency": 37.31,
                "stderr": 0.682,
                "cost_per_test": 0.020998,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "google/gemini-3.6-flash": {
                "accuracy": 93.457,
                "latency": 8.325,
                "stderr": 0.686,
                "cost_per_test": 0.015989,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 93.457,
                "latency": 561.23,
                "stderr": 0.701,
                "cost_per_test": 0.038088,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 93.38,
                "latency": 42.839,
                "stderr": 0.69,
                "cost_per_test": 0.023862,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "meta/muse_spark_1_2": {
                "accuracy": 93.38,
                "latency": 47.748,
                "stderr": 0.69,
                "cost_per_test": 0.010858,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "anthropic/claude-opus-4-7": {
                "accuracy": 93.38,
                "latency": 90.591,
                "stderr": 0.69,
                "cost_per_test": 0.053061,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3.8-max": {
                "accuracy": 93.303,
                "latency": 60.36,
                "stderr": 0.694,
                "cost_per_test": 0.01678,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/gpt-5.6-sol": {
                "accuracy": 92.764,
                "latency": 13.717,
                "stderr": 0.719,
                "cost_per_test": 0.032534,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 92.764,
                "latency": 55.516,
                "stderr": 0.719,
                "cost_per_test": 0.010983,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "alibaba/qwen3.5-plus-thinking": {
                "accuracy": 92.764,
                "latency": 81.557,
                "stderr": 0.719,
                "cost_per_test": 0.010281,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "meta/muse_spark": {
                "accuracy": 92.687,
                "latency": 50.858,
                "stderr": 0.722,
                "cost_per_test": 0.001656,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "kimi/kimi-k3": {
                "accuracy": 92.61,
                "latency": 47.586,
                "stderr": 0.726,
                "cost_per_test": 0.02405,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 92.61,
                "latency": 64.215,
                "stderr": 1.234,
                "cost_per_test": 0.08129,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "deepseek/deepseek-v4-pro-0813": {
                "accuracy": 92.456,
                "latency": 71.374,
                "stderr": 0.733,
                "cost_per_test": 0.014755,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "deepseek/deepseek-v4-flash-0731": {
                "accuracy": 92.379,
                "latency": 17.441,
                "stderr": 0.736,
                "cost_per_test": 0.002804,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "zai/glm-5.1": {
                "accuracy": 92.302,
                "latency": 117.665,
                "stderr": 0.74,
                "cost_per_test": 0.018672,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "grok/grok-4.3": {
                "accuracy": 92.302,
                "latency": 284.034,
                "stderr": 0.74,
                "cost_per_test": 0.011467,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "nvidia/nemotron-3-ultra-550b-a55b": {
                "accuracy": 92.225,
                "latency": 46.02,
                "stderr": 0.743,
                "cost_per_test": null,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Nvidia",
                "harness": null
            },
            "zai/glm-5.2": {
                "accuracy": 92.225,
                "latency": 63.91,
                "stderr": 0.743,
                "cost_per_test": 0.017374,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "thinkingmachines/inkling-small": {
                "accuracy": 92.225,
                "latency": 65.578,
                "stderr": 0.743,
                "cost_per_test": 0.006179,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 262000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "grok/grok-4.6": {
                "accuracy": 92.148,
                "latency": 47.373,
                "stderr": 0.746,
                "cost_per_test": 0.016782,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "zai/glm-5.3": {
                "accuracy": 92.148,
                "latency": 52.003,
                "stderr": 0.746,
                "cost_per_test": 0.014614,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 92.071,
                "latency": 186.541,
                "stderr": 0.753,
                "cost_per_test": 0.022914,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 91.994,
                "latency": 10.656,
                "stderr": 0.753,
                "cost_per_test": 0.001305,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "thinkingmachines/inkling": {
                "accuracy": 91.994,
                "latency": 66.021,
                "stderr": 0.753,
                "cost_per_test": 0.041044,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "grok/grok-4.5": {
                "accuracy": 91.994,
                "latency": 257.307,
                "stderr": 0.753,
                "cost_per_test": 0.017827,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "zai/glm-5.3-flash": {
                "accuracy": 91.84,
                "latency": 41.786,
                "stderr": 0.76,
                "cost_per_test": 0.001007,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "openai/gpt-5.6-luna": {
                "accuracy": 91.763,
                "latency": 14.975,
                "stderr": 0.815,
                "cost_per_test": 0.002594,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 91.686,
                "latency": 43.672,
                "stderr": 1.171,
                "cost_per_test": 0.050012,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "zai/glm-5-thinking": {
                "accuracy": 91.686,
                "latency": 109.966,
                "stderr": 0.772,
                "cost_per_test": 0.01562,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "anthropic/claude-sonnet-5": {
                "accuracy": 91.532,
                "latency": 8.615,
                "stderr": 0.871,
                "cost_per_test": 0.018511,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 91.532,
                "latency": 40.216,
                "stderr": 0.772,
                "cost_per_test": 0.024257,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 91.455,
                "latency": 11.019,
                "stderr": 0.776,
                "cost_per_test": 0.014481,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "alibaba/qwen3.5-flash": {
                "accuracy": 91.378,
                "latency": 31.736,
                "stderr": 0.779,
                "cost_per_test": 0.001891,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 91.378,
                "latency": 34.187,
                "stderr": 0.779,
                "cost_per_test": 0.039451,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 91.301,
                "latency": 56.633,
                "stderr": 0.782,
                "cost_per_test": 0.011735,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "alibaba/qwen3.8-27b": {
                "accuracy": 91.301,
                "latency": 57.97,
                "stderr": 0.782,
                "cost_per_test": 0.009675,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 91.224,
                "latency": 19.87,
                "stderr": 0.931,
                "cost_per_test": 0.041984,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "xiaomi/mimo-v2.5-pro": {
                "accuracy": 91.224,
                "latency": 93.164,
                "stderr": 1.052,
                "cost_per_test": 0.003086,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "alibaba/qwen3-max-2026-01-23": {
                "accuracy": 91.224,
                "latency": 136.647,
                "stderr": 0.785,
                "cost_per_test": 0.032391,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/gpt-5.6-terra": {
                "accuracy": 91.07,
                "latency": 7.846,
                "stderr": 0.791,
                "cost_per_test": 0.008051,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "minimax/MiniMax-M3": {
                "accuracy": 90.839,
                "latency": 50.009,
                "stderr": 0.847,
                "cost_per_test": 0.006439,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 512000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "fireworks/qwen3-235b-a22b": {
                "accuracy": 90.762,
                "latency": 57.666,
                "stderr": 0.818,
                "cost_per_test": 0.003131,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 90.685,
                "latency": 23.472,
                "stderr": 0.806,
                "cost_per_test": 0.022039,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 90.685,
                "latency": 44.486,
                "stderr": 0.806,
                "cost_per_test": 0.027143,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3-max": {
                "accuracy": 90.608,
                "latency": 0.0,
                "stderr": 0.809,
                "cost_per_test": 0.007626,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "google/gemini-3.5-flash-lite": {
                "accuracy": 90.531,
                "latency": 5.748,
                "stderr": 0.812,
                "cost_per_test": 0.005889,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 90.531,
                "latency": 16.319,
                "stderr": 0.812,
                "cost_per_test": 0.003314,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "xiaomi/mimo-v2.5": {
                "accuracy": 90.223,
                "latency": 36.498,
                "stderr": 0.835,
                "cost_per_test": 0.001022,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "zai/glm-4.6": {
                "accuracy": 90.223,
                "latency": 56.515,
                "stderr": 0.833,
                "cost_per_test": 0.007233,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 122880,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "grok/grok-4-0709": {
                "accuracy": 90.069,
                "latency": 120.658,
                "stderr": 0.83,
                "cost_per_test": 0.069894,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/o3-2025-04-16": {
                "accuracy": 89.915,
                "latency": 17.865,
                "stderr": 0.835,
                "cost_per_test": 0.013916,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-4-1-20250805-thinking": {
                "accuracy": 89.915,
                "latency": 34.366,
                "stderr": 0.841,
                "cost_per_test": 0.154028,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 89.915,
                "latency": 47.367,
                "stderr": 0.989,
                "cost_per_test": 0.282054,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101": {
                "accuracy": 89.838,
                "latency": 7.714,
                "stderr": 0.902,
                "cost_per_test": 0.050949,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514-thinking": {
                "accuracy": 89.838,
                "latency": 59.061,
                "stderr": 0.844,
                "cost_per_test": 0.060425,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-2.5-pro-exp-03-25": {
                "accuracy": 89.761,
                "latency": 21.681,
                "stderr": 0.841,
                "cost_per_test": 0.009687,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025": {
                "accuracy": 89.53,
                "latency": 8.471,
                "stderr": 0.849,
                "cost_per_test": 0.001752,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 89.53,
                "latency": 30.45,
                "stderr": 0.849,
                "cost_per_test": 0.001349,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "fireworks/deepseek-r1": {
                "accuracy": 89.53,
                "latency": 38.631,
                "stderr": 1.265,
                "cost_per_test": 0.016036,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "minimax/MiniMax-M2.1": {
                "accuracy": 89.53,
                "latency": 52.388,
                "stderr": 0.849,
                "cost_per_test": 0.003995,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "fireworks/deepseek-v3p2-thinking": {
                "accuracy": 89.53,
                "latency": 112.283,
                "stderr": 0.849,
                "cost_per_test": 0.001525,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "zai/glm-4.7": {
                "accuracy": 89.53,
                "latency": 135.407,
                "stderr": 0.852,
                "cost_per_test": 0.012007,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 89.453,
                "latency": 17.887,
                "stderr": 0.852,
                "cost_per_test": 0.005539,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "ant/ling-3.0-flash-2607": {
                "accuracy": 89.376,
                "latency": 25.989,
                "stderr": 0.871,
                "cost_per_test": 0.001273,
                "temperature": 0.6,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "anthropic/claude-opus-4-1-20250805": {
                "accuracy": 88.915,
                "latency": 18.345,
                "stderr": 0.871,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-3-mini-fast-high-reasoning": {
                "accuracy": 88.838,
                "latency": 27.838,
                "stderr": 0.874,
                "cost_per_test": 0.002409,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025-thinking": {
                "accuracy": 88.684,
                "latency": 11.403,
                "stderr": 0.879,
                "cost_per_test": 0.001754,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3-max-preview": {
                "accuracy": 88.376,
                "latency": 53.018,
                "stderr": 0.889,
                "cost_per_test": 0.00767,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "fireworks/deepseek-v3p2": {
                "accuracy": 88.299,
                "latency": 31.74,
                "stderr": 0.945,
                "cost_per_test": 0.00037,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "anthropic/claude-opus-4-20250514": {
                "accuracy": 87.991,
                "latency": 10.681,
                "stderr": 0.902,
                "cost_per_test": 0.050261,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 87.991,
                "latency": 67.562,
                "stderr": 0.909,
                "cost_per_test": 0.005466,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "grok/grok-3": {
                "accuracy": 87.914,
                "latency": 11.498,
                "stderr": 0.904,
                "cost_per_test": 0.01956,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 87.76,
                "latency": 10.012,
                "stderr": 0.974,
                "cost_per_test": 0.000492,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/o3-mini-2025-01-31": {
                "accuracy": 87.76,
                "latency": 22.351,
                "stderr": 0.909,
                "cost_per_test": 0.013089,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-3-7-sonnet-20250219-thinking": {
                "accuracy": 87.76,
                "latency": 44.271,
                "stderr": 0.909,
                "cost_per_test": 0.058227,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/o1-2024-12-17": {
                "accuracy": 87.529,
                "latency": 33.158,
                "stderr": 0.917,
                "cost_per_test": 0.131873,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "minimax/MiniMax-M2.5": {
                "accuracy": 87.529,
                "latency": 55.012,
                "stderr": 0.922,
                "cost_per_test": 0.006352,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "openai/o4-mini-2025-04-16": {
                "accuracy": 87.375,
                "latency": 10.661,
                "stderr": 0.922,
                "cost_per_test": 0.005527,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-4.5": {
                "accuracy": 87.144,
                "latency": 96.592,
                "stderr": 0.929,
                "cost_per_test": 0.013783,
                "temperature": 0.6,
                "top_p": 1.0,
                "max_output_tokens": 81920,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "fireworks/gpt-oss-120b": {
                "accuracy": 86.913,
                "latency": 18.674,
                "stderr": 0.945,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "grok/grok-4-fast-reasoning": {
                "accuracy": 86.836,
                "latency": 9.383,
                "stderr": 0.938,
                "cost_per_test": 0.001892,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "fireworks/llama4-maverick-instruct-basic": {
                "accuracy": 86.605,
                "latency": 5.019,
                "stderr": 0.945,
                "cost_per_test": 0.000682,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "mistralai/mistral-medium-3.5": {
                "accuracy": 86.528,
                "latency": 71.364,
                "stderr": 0.947,
                "cost_per_test": 0.056352,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "grok/grok-3-mini-fast-low-reasoning": {
                "accuracy": 86.451,
                "latency": 8.814,
                "stderr": 0.95,
                "cost_per_test": 0.001151,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "low",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "kimi/kimi-k2-thinking": {
                "accuracy": 86.451,
                "latency": 98.509,
                "stderr": 1.014,
                "cost_per_test": 0.0022,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "mistralai/mistral-large-2512": {
                "accuracy": 86.374,
                "latency": 14.859,
                "stderr": 1.213,
                "cost_per_test": 0.001605,
                "temperature": 0.065,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "fireworks/deepseek-v3-0324": {
                "accuracy": 86.22,
                "latency": 22.957,
                "stderr": 1.016,
                "cost_per_test": 0.001517,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025-thinking": {
                "accuracy": 86.066,
                "latency": 10.632,
                "stderr": 0.961,
                "cost_per_test": 0.000356,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514": {
                "accuracy": 85.758,
                "latency": 10.151,
                "stderr": 0.97,
                "cost_per_test": 0.010341,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "together/moonshotai/Kimi-K2-Instruct": {
                "accuracy": 85.604,
                "latency": 23.06,
                "stderr": 0.974,
                "cost_per_test": 0.002917,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "openai/gpt-5-nano-2025-08-07": {
                "accuracy": 85.527,
                "latency": 22.595,
                "stderr": 0.978,
                "cost_per_test": 0.002019,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 85.45,
                "latency": 34.158,
                "stderr": 1.299,
                "cost_per_test": 0.026522,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-4.1-2025-04-14": {
                "accuracy": 84.834,
                "latency": 8.881,
                "stderr": 0.995,
                "cost_per_test": 0.007688,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025": {
                "accuracy": 84.45,
                "latency": 3.727,
                "stderr": 1.005,
                "cost_per_test": 0.000677,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-3-7-sonnet-20250219": {
                "accuracy": 84.296,
                "latency": 6.592,
                "stderr": 1.01,
                "cost_per_test": 0.009941,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-2.0-flash-001": {
                "accuracy": 83.911,
                "latency": 5.669,
                "stderr": 1.019,
                "cost_per_test": 0.000517,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-4.1-mini-2025-04-14": {
                "accuracy": 83.603,
                "latency": 4.687,
                "stderr": 1.027,
                "cost_per_test": 0.001481,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "fireworks/gpt-oss-20b": {
                "accuracy": 82.602,
                "latency": 22.214,
                "stderr": 1.154,
                "cost_per_test": 0.00112,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "google/gemini-1.5-pro-002": {
                "accuracy": 81.293,
                "latency": 4.62,
                "stderr": 1.084,
                "cost_per_test": 0.003023,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "mistralai/mistral-medium-2505": {
                "accuracy": 80.216,
                "latency": 14.915,
                "stderr": 1.109,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "grok/grok-2-1212": {
                "accuracy": 79.523,
                "latency": 13.432,
                "stderr": 1.12,
                "cost_per_test": 0.007152,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 0,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-3-5-sonnet-20241022": {
                "accuracy": 78.214,
                "latency": 6.246,
                "stderr": 1.145,
                "cost_per_test": 0.008225,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-4-1-fast-non-reasoning": {
                "accuracy": 78.06,
                "latency": 4.979,
                "stderr": 1.148,
                "cost_per_test": 0.00025,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "fireworks/deepseek-v3": {
                "accuracy": 77.213,
                "latency": 7.147,
                "stderr": 1.164,
                "cost_per_test": 0.001282,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "together/meta-llama/Llama-4-Scout-17B-16E-Instruct": {
                "accuracy": 76.828,
                "latency": 8.434,
                "stderr": 1.171,
                "cost_per_test": 0.000539,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "mistralai/magistral-small-2509": {
                "accuracy": 75.905,
                "latency": 25.587,
                "stderr": 1.187,
                "cost_per_test": 0.004659,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "mistralai/magistral-medium-2509": {
                "accuracy": 75.828,
                "latency": 38.255,
                "stderr": 1.188,
                "cost_per_test": 0.016465,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "grok/grok-4-fast-non-reasoning": {
                "accuracy": 75.751,
                "latency": 2.071,
                "stderr": 1.189,
                "cost_per_test": 0.001043,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-4o-2024-08-06": {
                "accuracy": 75.366,
                "latency": 12.704,
                "stderr": 1.196,
                "cost_per_test": 0.008408,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-4o-2024-11-20": {
                "accuracy": 74.211,
                "latency": 12.097,
                "stderr": 1.271,
                "cost_per_test": 0.008214,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "poolside/laguna-xs.2": {
                "accuracy": 72.209,
                "latency": 66.832,
                "stderr": 1.342,
                "cost_per_test": 0.000914,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "mistralai/mistral-large-2411": {
                "accuracy": 72.055,
                "latency": 10.883,
                "stderr": 1.245,
                "cost_per_test": 0.005755,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "poolside/laguna-m.1": {
                "accuracy": 71.44,
                "latency": 145.587,
                "stderr": 1.262,
                "cost_per_test": 0.002875,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "google/gemini-1.5-flash-002": {
                "accuracy": 71.209,
                "latency": 2.297,
                "stderr": 1.256,
                "cost_per_test": 0.000188,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "together/meta-llama/Llama-3.3-70B-Instruct-Turbo": {
                "accuracy": 69.284,
                "latency": 4.451,
                "stderr": 1.28,
                "cost_per_test": 0.001542,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "cohere/command-a-03-2025": {
                "accuracy": 69.284,
                "latency": 15.203,
                "stderr": 1.298,
                "cost_per_test": 0.008877,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 8000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "openai/gpt-4.1-nano-2025-04-14": {
                "accuracy": 69.053,
                "latency": 2.658,
                "stderr": 1.299,
                "cost_per_test": 0.000363,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "mistralai/mistral-small-2402": {
                "accuracy": 68.976,
                "latency": 7.145,
                "stderr": 1.283,
                "cost_per_test": 0.00061,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "mistralai/mistral-small-2503": {
                "accuracy": 68.206,
                "latency": 5.861,
                "stderr": 1.292,
                "cost_per_test": 0.000262,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "together/langston/nim/nvidia/llama-3.3-nemotron-super-49b-v1-42e84561": {
                "accuracy": 66.975,
                "latency": 18.493,
                "stderr": 1.328,
                "cost_per_test": null,
                "temperature": 0.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "together/langston/nim/nvidia/llama-3.3-nemotron-super-49b-v1-42e84561-thinking": {
                "accuracy": 64.896,
                "latency": 50.202,
                "stderr": 1.331,
                "cost_per_test": null,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "openai/gpt-4o-mini-2024-07-18": {
                "accuracy": 64.434,
                "latency": 7.722,
                "stderr": 1.328,
                "cost_per_test": 0.000508,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-3-5-haiku-20241022": {
                "accuracy": 61.432,
                "latency": 5.691,
                "stderr": 1.351,
                "cost_per_test": 0.00214,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "ai21labs/jamba-large-1.6": {
                "accuracy": 52.733,
                "latency": 15.319,
                "stderr": 1.387,
                "cost_per_test": 0.005276,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "cohere/command-r-plus": {
                "accuracy": 34.642,
                "latency": 4.669,
                "stderr": 1.32,
                "cost_per_test": 0.005526,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 4000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "ai21labs/jamba-mini-1.6": {
                "accuracy": 30.023,
                "latency": 3.412,
                "stderr": 1.27,
                "cost_per_test": 0.000426,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            }
        },
        "psychology": {
            "anthropic/claude-fable-5-1": {
                "accuracy": 93.734,
                "latency": 12.25,
                "stderr": 0.858,
                "cost_per_test": 0.057011,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-5": {
                "accuracy": 92.481,
                "latency": 6.875,
                "stderr": 0.933,
                "cost_per_test": 0.019329,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-fable-5": {
                "accuracy": 92.231,
                "latency": 13.228,
                "stderr": 0.948,
                "cost_per_test": 0.04971,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 91.98,
                "latency": 15.965,
                "stderr": 0.961,
                "cost_per_test": 0.015685,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.8-flash": {
                "accuracy": 91.479,
                "latency": 5.243,
                "stderr": 0.988,
                "cost_per_test": 0.013425,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-4-7": {
                "accuracy": 91.228,
                "latency": 37.575,
                "stderr": 1.045,
                "cost_per_test": 0.028932,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.7-flash": {
                "accuracy": 91.103,
                "latency": 2.915,
                "stderr": 1.008,
                "cost_per_test": 0.006702,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 90.977,
                "latency": 16.858,
                "stderr": 1.014,
                "cost_per_test": 0.024357,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-4.6": {
                "accuracy": 90.852,
                "latency": 29.715,
                "stderr": 1.021,
                "cost_per_test": 0.008823,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-3.5-flash": {
                "accuracy": 90.351,
                "latency": 8.646,
                "stderr": 1.045,
                "cost_per_test": 0.015991,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-4.5": {
                "accuracy": 90.226,
                "latency": 341.948,
                "stderr": 1.051,
                "cost_per_test": 0.00786,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-opus-4-8": {
                "accuracy": 89.975,
                "latency": 13.084,
                "stderr": 1.063,
                "cost_per_test": 0.031378,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3.7-max": {
                "accuracy": 89.975,
                "latency": 19.369,
                "stderr": 1.063,
                "cost_per_test": 0.010724,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 89.85,
                "latency": 18.714,
                "stderr": 1.767,
                "cost_per_test": 0.017434,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.6-flash": {
                "accuracy": 89.474,
                "latency": 5.097,
                "stderr": 1.086,
                "cost_per_test": 0.00853,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-4-1-20250805": {
                "accuracy": 89.348,
                "latency": 16.632,
                "stderr": 1.092,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.6-sol": {
                "accuracy": 89.223,
                "latency": 6.702,
                "stderr": 1.098,
                "cost_per_test": 0.016633,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 89.223,
                "latency": 89.709,
                "stderr": 1.211,
                "cost_per_test": 0.009941,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 88.847,
                "latency": 20.086,
                "stderr": 1.114,
                "cost_per_test": 0.010053,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101": {
                "accuracy": 88.722,
                "latency": 7.824,
                "stderr": 1.125,
                "cost_per_test": 0.042011,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-sonnet-5": {
                "accuracy": 88.722,
                "latency": 27.27,
                "stderr": 1.239,
                "cost_per_test": 0.045438,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-4-1-20250805-thinking": {
                "accuracy": 88.471,
                "latency": 19.271,
                "stderr": 1.131,
                "cost_per_test": 0.080786,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 88.471,
                "latency": 24.149,
                "stderr": 1.131,
                "cost_per_test": 0.017305,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3.8-max": {
                "accuracy": 88.471,
                "latency": 33.739,
                "stderr": 1.131,
                "cost_per_test": 0.010432,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "kimi/kimi-k3": {
                "accuracy": 88.346,
                "latency": 25.073,
                "stderr": 1.136,
                "cost_per_test": 0.009787,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "anthropic/claude-opus-4-20250514": {
                "accuracy": 88.221,
                "latency": 9.928,
                "stderr": 1.141,
                "cost_per_test": 0.039244,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 88.221,
                "latency": 32.451,
                "stderr": 1.187,
                "cost_per_test": 0.111833,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "minimax/MiniMax-M2.1": {
                "accuracy": 87.97,
                "latency": 22.426,
                "stderr": 1.157,
                "cost_per_test": 0.001849,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 87.97,
                "latency": 492.984,
                "stderr": 1.167,
                "cost_per_test": 0.009001,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 87.845,
                "latency": 24.507,
                "stderr": 1.192,
                "cost_per_test": 0.022662,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 87.845,
                "latency": 34.537,
                "stderr": 1.474,
                "cost_per_test": 0.032734,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4.3": {
                "accuracy": 87.719,
                "latency": 384.867,
                "stderr": 1.162,
                "cost_per_test": 0.005471,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 87.469,
                "latency": 22.032,
                "stderr": 1.172,
                "cost_per_test": 0.014089,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 87.469,
                "latency": 33.387,
                "stderr": 1.172,
                "cost_per_test": 0.006021,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 87.343,
                "latency": 28.875,
                "stderr": 1.339,
                "cost_per_test": 0.021554,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "meta/muse_spark_1_1": {
                "accuracy": 87.218,
                "latency": 15.16,
                "stderr": 1.187,
                "cost_per_test": 0.007738,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "zai/glm-5.1": {
                "accuracy": 87.218,
                "latency": 26.174,
                "stderr": 1.182,
                "cost_per_test": 0.004468,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "deepseek/deepseek-v4-pro-0813": {
                "accuracy": 87.218,
                "latency": 28.213,
                "stderr": 1.187,
                "cost_per_test": 0.00544,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 87.093,
                "latency": 11.907,
                "stderr": 1.187,
                "cost_per_test": 0.009152,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3.5-plus-thinking": {
                "accuracy": 86.967,
                "latency": 55.554,
                "stderr": 1.192,
                "cost_per_test": 0.006254,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "zai/glm-5.3-flash": {
                "accuracy": 86.717,
                "latency": 13.493,
                "stderr": 1.201,
                "cost_per_test": 0.000313,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "meta/muse_spark_1_2": {
                "accuracy": 86.591,
                "latency": 27.427,
                "stderr": 1.211,
                "cost_per_test": 0.00688,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "thinkingmachines/inkling": {
                "accuracy": 86.591,
                "latency": 39.249,
                "stderr": 1.206,
                "cost_per_test": 0.019293,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "openai/o3-2025-04-16": {
                "accuracy": 86.466,
                "latency": 11.696,
                "stderr": 1.211,
                "cost_per_test": 0.007081,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "meta/muse_spark": {
                "accuracy": 86.466,
                "latency": 31.285,
                "stderr": 1.211,
                "cost_per_test": 0.001246,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "openai/gpt-5.6-terra": {
                "accuracy": 86.341,
                "latency": 4.315,
                "stderr": 1.216,
                "cost_per_test": 0.004611,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-5.2": {
                "accuracy": 86.216,
                "latency": 30.286,
                "stderr": 1.22,
                "cost_per_test": 0.007257,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 86.09,
                "latency": 15.183,
                "stderr": 1.248,
                "cost_per_test": 0.015346,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 86.09,
                "latency": 23.815,
                "stderr": 1.225,
                "cost_per_test": 0.00459,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "xiaomi/mimo-v2.5-pro": {
                "accuracy": 86.09,
                "latency": 28.102,
                "stderr": 1.769,
                "cost_per_test": 0.000782,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "zai/glm-5.3": {
                "accuracy": 86.09,
                "latency": 32.444,
                "stderr": 1.225,
                "cost_per_test": 0.008575,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025": {
                "accuracy": 85.714,
                "latency": 4.655,
                "stderr": 1.239,
                "cost_per_test": 0.000934,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514": {
                "accuracy": 85.714,
                "latency": 7.25,
                "stderr": 1.239,
                "cost_per_test": 0.008135,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 85.714,
                "latency": 19.037,
                "stderr": 1.239,
                "cost_per_test": 0.000661,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.5-flash-lite": {
                "accuracy": 85.589,
                "latency": 3.199,
                "stderr": 1.243,
                "cost_per_test": 0.002565,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "deepseek/deepseek-v4-flash-0731": {
                "accuracy": 85.589,
                "latency": 8.214,
                "stderr": 1.243,
                "cost_per_test": 0.001056,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 85.589,
                "latency": 12.005,
                "stderr": 1.243,
                "cost_per_test": 0.00056,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "zai/glm-5-thinking": {
                "accuracy": 85.589,
                "latency": 50.749,
                "stderr": 1.243,
                "cost_per_test": 0.006019,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 85.338,
                "latency": 7.022,
                "stderr": 1.252,
                "cost_per_test": 0.007685,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514-thinking": {
                "accuracy": 85.338,
                "latency": 22.345,
                "stderr": 1.252,
                "cost_per_test": 0.02236,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-2.5-pro-exp-03-25": {
                "accuracy": 85.088,
                "latency": 13.171,
                "stderr": 1.261,
                "cost_per_test": 0.007016,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.6-luna": {
                "accuracy": 84.837,
                "latency": 10.477,
                "stderr": 1.315,
                "cost_per_test": 0.001399,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3-max-preview": {
                "accuracy": 84.837,
                "latency": 25.18,
                "stderr": 1.27,
                "cost_per_test": 0.003869,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "grok/grok-4-0709": {
                "accuracy": 84.837,
                "latency": 39.89,
                "stderr": 1.27,
                "cost_per_test": 0.022401,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "alibaba/qwen3.8-27b": {
                "accuracy": 84.837,
                "latency": 85.23,
                "stderr": 1.27,
                "cost_per_test": 0.004891,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "alibaba/qwen3-max": {
                "accuracy": 84.812,
                "latency": 0.0,
                "stderr": 1.251,
                "cost_per_test": 0.00375,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025-thinking": {
                "accuracy": 84.586,
                "latency": 8.534,
                "stderr": 1.278,
                "cost_per_test": 0.00095,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "nvidia/nemotron-3-ultra-550b-a55b": {
                "accuracy": 84.586,
                "latency": 13.753,
                "stderr": 1.278,
                "cost_per_test": null,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Nvidia",
                "harness": null
            },
            "fireworks/deepseek-v3p2": {
                "accuracy": 84.586,
                "latency": 41.38,
                "stderr": 1.404,
                "cost_per_test": 0.000171,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "fireworks/deepseek-v3p2-thinking": {
                "accuracy": 84.461,
                "latency": 58.865,
                "stderr": 1.282,
                "cost_per_test": 0.000616,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 84.211,
                "latency": 13.894,
                "stderr": 1.299,
                "cost_per_test": 0.002023,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/o1-2024-12-17": {
                "accuracy": 84.211,
                "latency": 17.866,
                "stderr": 1.291,
                "cost_per_test": 0.074227,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "thinkingmachines/inkling-small": {
                "accuracy": 83.96,
                "latency": 36.831,
                "stderr": 1.299,
                "cost_per_test": 0.002473,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 262000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "alibaba/qwen3-max-2026-01-23": {
                "accuracy": 83.96,
                "latency": 76.522,
                "stderr": 1.299,
                "cost_per_test": 0.016928,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "zai/glm-4.7": {
                "accuracy": 83.584,
                "latency": 64.321,
                "stderr": 1.315,
                "cost_per_test": 0.003926,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "openai/gpt-4.1-2025-04-14": {
                "accuracy": 83.333,
                "latency": 2.093,
                "stderr": 1.319,
                "cost_per_test": 0.003408,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4-fast-reasoning": {
                "accuracy": 83.333,
                "latency": 3.847,
                "stderr": 1.319,
                "cost_per_test": 0.001148,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "minimax/MiniMax-M3": {
                "accuracy": 83.333,
                "latency": 25.992,
                "stderr": 1.323,
                "cost_per_test": 0.002098,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 512000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "anthropic/claude-3-7-sonnet-20250219": {
                "accuracy": 83.208,
                "latency": 5.9,
                "stderr": 1.323,
                "cost_per_test": 0.007899,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3.5-flash": {
                "accuracy": 83.208,
                "latency": 24.811,
                "stderr": 1.323,
                "cost_per_test": 0.001243,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "zai/glm-4.6": {
                "accuracy": 83.208,
                "latency": 39.645,
                "stderr": 1.327,
                "cost_per_test": 0.004307,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 122880,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "kimi/kimi-k2-thinking": {
                "accuracy": 83.208,
                "latency": 64.574,
                "stderr": 1.35,
                "cost_per_test": 0.001324,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "openai/gpt-4o-2024-11-20": {
                "accuracy": 82.957,
                "latency": 6.706,
                "stderr": 1.35,
                "cost_per_test": 0.005454,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "mistralai/mistral-medium-3.5": {
                "accuracy": 82.957,
                "latency": 52.096,
                "stderr": 1.331,
                "cost_per_test": 0.022867,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "ant/ling-3.0-flash-2607": {
                "accuracy": 82.832,
                "latency": 5.165,
                "stderr": 1.335,
                "cost_per_test": 0.00018,
                "temperature": 0.6,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "fireworks/deepseek-r1": {
                "accuracy": 82.707,
                "latency": 16.482,
                "stderr": 1.603,
                "cost_per_test": 0.007569,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "together/moonshotai/Kimi-K2-Instruct": {
                "accuracy": 82.456,
                "latency": 11.527,
                "stderr": 1.346,
                "cost_per_test": 0.001865,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 82.456,
                "latency": 14.49,
                "stderr": 1.346,
                "cost_per_test": 0.003107,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "fireworks/qwen3-235b-a22b": {
                "accuracy": 82.206,
                "latency": 42.376,
                "stderr": 1.369,
                "cost_per_test": 0.001254,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "anthropic/claude-3-7-sonnet-20250219-thinking": {
                "accuracy": 82.08,
                "latency": 18.433,
                "stderr": 1.358,
                "cost_per_test": 0.021302,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/o4-mini-2025-04-16": {
                "accuracy": 81.83,
                "latency": 7.088,
                "stderr": 1.365,
                "cost_per_test": 0.004056,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-2-1212": {
                "accuracy": 81.704,
                "latency": 3.194,
                "stderr": 1.369,
                "cost_per_test": 0.003577,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 0,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "zai/glm-4.5": {
                "accuracy": 81.579,
                "latency": 70.368,
                "stderr": 1.372,
                "cost_per_test": 0.005607,
                "temperature": 0.6,
                "top_p": 1.0,
                "max_output_tokens": 81920,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "fireworks/deepseek-v3-0324": {
                "accuracy": 81.454,
                "latency": 10.137,
                "stderr": 1.376,
                "cost_per_test": 0.001074,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "anthropic/claude-3-5-sonnet-20241022": {
                "accuracy": 81.328,
                "latency": 6.511,
                "stderr": 1.379,
                "cost_per_test": 0.00733,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "fireworks/llama4-maverick-instruct-basic": {
                "accuracy": 81.078,
                "latency": 4.688,
                "stderr": 1.387,
                "cost_per_test": 0.000597,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "grok/grok-3-mini-fast-low-reasoning": {
                "accuracy": 81.078,
                "latency": 5.093,
                "stderr": 1.387,
                "cost_per_test": 0.000756,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "low",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025-thinking": {
                "accuracy": 81.078,
                "latency": 6.742,
                "stderr": 1.387,
                "cost_per_test": 0.000205,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-3-mini-fast-high-reasoning": {
                "accuracy": 81.078,
                "latency": 7.35,
                "stderr": 1.387,
                "cost_per_test": 0.000978,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "xiaomi/mimo-v2.5": {
                "accuracy": 81.078,
                "latency": 15.633,
                "stderr": 1.387,
                "cost_per_test": 0.000348,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "mistralai/magistral-medium-2509": {
                "accuracy": 80.952,
                "latency": 17.485,
                "stderr": 1.397,
                "cost_per_test": 0.008589,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "openai/o3-mini-2025-01-31": {
                "accuracy": 80.827,
                "latency": 14.281,
                "stderr": 1.394,
                "cost_per_test": 0.008321,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-4o-2024-08-06": {
                "accuracy": 80.702,
                "latency": 6.961,
                "stderr": 1.4,
                "cost_per_test": 0.005488,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-1.5-pro-002": {
                "accuracy": 80.576,
                "latency": 1.803,
                "stderr": 1.4,
                "cost_per_test": 0.001761,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-2.0-flash-001": {
                "accuracy": 80.576,
                "latency": 1.826,
                "stderr": 1.404,
                "cost_per_test": 0.000194,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "mistralai/mistral-large-2512": {
                "accuracy": 80.576,
                "latency": 10.746,
                "stderr": 1.754,
                "cost_per_test": 0.001079,
                "temperature": 0.065,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "grok/grok-4-1-fast-non-reasoning": {
                "accuracy": 80.451,
                "latency": 1.854,
                "stderr": 1.404,
                "cost_per_test": 0.000151,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "mistralai/mistral-medium-2505": {
                "accuracy": 80.326,
                "latency": 4.791,
                "stderr": 1.459,
                "cost_per_test": 0.000878,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025": {
                "accuracy": 79.825,
                "latency": 2.191,
                "stderr": 1.421,
                "cost_per_test": 0.000338,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-3": {
                "accuracy": 79.825,
                "latency": 11.534,
                "stderr": 1.421,
                "cost_per_test": 0.012964,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 79.825,
                "latency": 14.232,
                "stderr": 1.76,
                "cost_per_test": 0.008931,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "minimax/MiniMax-M2.5": {
                "accuracy": 79.073,
                "latency": 25.216,
                "stderr": 1.44,
                "cost_per_test": 0.002921,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "openai/gpt-4.1-mini-2025-04-14": {
                "accuracy": 78.822,
                "latency": 2.333,
                "stderr": 1.446,
                "cost_per_test": 0.000888,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "fireworks/deepseek-v3": {
                "accuracy": 78.822,
                "latency": 3.148,
                "stderr": 1.446,
                "cost_per_test": 0.001019,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 77.945,
                "latency": 23.345,
                "stderr": 1.485,
                "cost_per_test": 0.001847,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "grok/grok-4-fast-non-reasoning": {
                "accuracy": 77.82,
                "latency": 1.579,
                "stderr": 1.471,
                "cost_per_test": 0.000896,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-5-nano-2025-08-07": {
                "accuracy": 77.82,
                "latency": 16.134,
                "stderr": 1.471,
                "cost_per_test": 0.001158,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "fireworks/gpt-oss-120b": {
                "accuracy": 77.82,
                "latency": 52.376,
                "stderr": 1.477,
                "cost_per_test": 0.001118,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "mistralai/mistral-large-2411": {
                "accuracy": 77.193,
                "latency": 4.051,
                "stderr": 1.485,
                "cost_per_test": 0.003469,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "together/langston/nim/nvidia/llama-3.3-nemotron-super-49b-v1-42e84561-thinking": {
                "accuracy": 77.068,
                "latency": 20.936,
                "stderr": 1.494,
                "cost_per_test": null,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "poolside/laguna-m.1": {
                "accuracy": 76.316,
                "latency": 62.013,
                "stderr": 1.524,
                "cost_per_test": 0.000713,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "together/meta-llama/Llama-3.3-70B-Instruct-Turbo": {
                "accuracy": 75.815,
                "latency": 4.183,
                "stderr": 1.516,
                "cost_per_test": 0.001251,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "together/meta-llama/Llama-4-Scout-17B-16E-Instruct": {
                "accuracy": 75.564,
                "latency": 4.182,
                "stderr": 1.521,
                "cost_per_test": 0.000414,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "poolside/laguna-xs.2": {
                "accuracy": 75.564,
                "latency": 26.139,
                "stderr": 1.719,
                "cost_per_test": 0.00027,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 74.812,
                "latency": 3.55,
                "stderr": 1.629,
                "cost_per_test": 0.000169,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-1.5-flash-002": {
                "accuracy": 74.687,
                "latency": 1.181,
                "stderr": 1.539,
                "cost_per_test": 0.000126,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "cohere/command-a-03-2025": {
                "accuracy": 74.687,
                "latency": 7.087,
                "stderr": 1.612,
                "cost_per_test": 0.005887,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 8000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "together/langston/nim/nvidia/llama-3.3-nemotron-super-49b-v1-42e84561": {
                "accuracy": 73.935,
                "latency": 18.032,
                "stderr": 1.681,
                "cost_per_test": null,
                "temperature": 0.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "mistralai/mistral-small-2503": {
                "accuracy": 73.183,
                "latency": 2.151,
                "stderr": 1.568,
                "cost_per_test": 0.000141,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "openai/gpt-4o-mini-2024-07-18": {
                "accuracy": 72.932,
                "latency": 2.934,
                "stderr": 1.573,
                "cost_per_test": 0.000277,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-3-5-haiku-20241022": {
                "accuracy": 72.431,
                "latency": 5.807,
                "stderr": 1.582,
                "cost_per_test": 0.001922,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-4.1-nano-2025-04-14": {
                "accuracy": 71.679,
                "latency": 1.642,
                "stderr": 1.603,
                "cost_per_test": 0.000228,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "fireworks/gpt-oss-20b": {
                "accuracy": 71.554,
                "latency": 42.466,
                "stderr": 1.661,
                "cost_per_test": 0.000615,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "mistralai/magistral-small-2509": {
                "accuracy": 71.429,
                "latency": 8.153,
                "stderr": 1.599,
                "cost_per_test": 0.001986,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "mistralai/mistral-small-2402": {
                "accuracy": 68.546,
                "latency": 2.667,
                "stderr": 1.644,
                "cost_per_test": 0.000347,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "ai21labs/jamba-large-1.6": {
                "accuracy": 64.912,
                "latency": 5.11,
                "stderr": 1.689,
                "cost_per_test": 0.00298,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "cohere/command-r-plus": {
                "accuracy": 60.025,
                "latency": 3.13,
                "stderr": 1.734,
                "cost_per_test": 0.004253,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 4000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "ai21labs/jamba-mini-1.6": {
                "accuracy": 26.942,
                "latency": 0.978,
                "stderr": 1.571,
                "cost_per_test": 0.000259,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            }
        }
    }
}