{
    "metadata": {
        "benchmark": "LegalBench",
        "slug": "legal_bench",
        "description": "Evaluating language models on a wide range of open source legal reasoning tasks.",
        "benchmark_id": "legal_bench",
        "family": "legal_bench",
        "version": "1",
        "updated": "2026-10-01",
        "dataset_type": "public",
        "industry": "legal",
        "tasks": {
            "overall": "Overall",
            "issue_tasks": "Issue Tasks",
            "rule_tasks": "Rule Tasks",
            "conclusion_tasks": "Conclusion Tasks",
            "interpretation_tasks": "Interpretation Tasks",
            "rhetoric_tasks": "Rhetoric Tasks"
        },
        "models": [
            "ai21labs/jamba-1.5-large",
            "ai21labs/jamba-1.5-mini",
            "ai21labs/jamba-large-1.6",
            "ai21labs/jamba-mini-1.6",
            "alibaba/qwen3-max",
            "alibaba/qwen3-max-preview",
            "alibaba/qwen3.5-flash",
            "alibaba/qwen3.5-plus-thinking",
            "alibaba/qwen3.6-plus",
            "alibaba/qwen3.7-max",
            "alibaba/qwen3.8-27b",
            "alibaba/qwen3.8-max",
            "ant/ling-3.0-flash-2607",
            "ant/ling-3.0-flash-af-rc3",
            "anthropic/claude-3-5-haiku-20241022",
            "anthropic/claude-3-7-sonnet-20250219",
            "anthropic/claude-3-7-sonnet-20250219-thinking",
            "anthropic/claude-fable-5",
            "anthropic/claude-fable-5-1",
            "anthropic/claude-haiku-4-5-20251001-thinking",
            "anthropic/claude-opus-4-1-20250805",
            "anthropic/claude-opus-4-20250514",
            "anthropic/claude-opus-4-5-20251101",
            "anthropic/claude-opus-4-5-20251101-thinking",
            "anthropic/claude-opus-4-6-thinking",
            "anthropic/claude-opus-4-7",
            "anthropic/claude-opus-4-8",
            "anthropic/claude-opus-5",
            "anthropic/claude-sonnet-4-20250514",
            "anthropic/claude-sonnet-4-20250514-thinking",
            "anthropic/claude-sonnet-4-5-20250929-thinking",
            "anthropic/claude-sonnet-4-6",
            "anthropic/claude-sonnet-5",
            "cohere/command-a-03-2025",
            "cohere/command-a-plus-05-2026",
            "cohere/command-r",
            "cohere/command-r-plus",
            "deepseek/deepseek-v4-flash-0731",
            "deepseek/deepseek-v4-pro",
            "deepseek/deepseek-v4-pro-0813",
            "deepseek/deepseek-v4.1-flash",
            "fireworks/deepseek-r1",
            "fireworks/deepseek-v3",
            "fireworks/deepseek-v3-0324",
            "fireworks/deepseek-v3p2",
            "fireworks/deepseek-v3p2-thinking",
            "fireworks/gpt-oss-120b",
            "fireworks/gpt-oss-20b",
            "fireworks/llama4-maverick-instruct-basic",
            "fireworks/qwen3-235b-a22b",
            "google/gemini-1.0-pro-002",
            "google/gemini-1.5-flash-002",
            "google/gemini-1.5-pro-002",
            "google/gemini-2.0-flash-001",
            "google/gemini-2.5-flash-lite-preview-09-2025",
            "google/gemini-2.5-flash-lite-preview-09-2025-thinking",
            "google/gemini-2.5-flash-preview-04-17",
            "google/gemini-2.5-flash-preview-09-2025",
            "google/gemini-2.5-flash-preview-09-2025-thinking",
            "google/gemini-2.5-pro-exp-03-25",
            "google/gemini-3-flash-preview",
            "google/gemini-3-pro-preview",
            "google/gemini-3.1-flash-lite-preview",
            "google/gemini-3.1-pro-preview",
            "google/gemini-3.5-flash",
            "google/gemini-3.5-flash-lite",
            "google/gemini-3.6-flash",
            "google/gemini-3.7-flash",
            "google/gemini-3.8-flash",
            "google/gemini-4-argon",
            "grok/grok-3",
            "grok/grok-3-mini-fast-high-reasoning",
            "grok/grok-3-mini-fast-low-reasoning",
            "grok/grok-4-0709",
            "grok/grok-4-1-fast-non-reasoning",
            "grok/grok-4-1-fast-reasoning",
            "grok/grok-4-fast-non-reasoning",
            "grok/grok-4-fast-reasoning",
            "grok/grok-4.20-0309-reasoning",
            "grok/grok-4.3",
            "grok/grok-4.5",
            "grok/grok-4.6",
            "grok/grok-4.7",
            "inception/mercury-2.5",
            "kimi/kimi-k2-thinking",
            "kimi/kimi-k2.6",
            "kimi/kimi-k3",
            "meta/muse_spark",
            "meta/muse_spark_1_1",
            "meta/muse_spark_1_2",
            "minimax/MiniMax-M2.1",
            "minimax/MiniMax-M2.5",
            "minimax/MiniMax-M2.7",
            "minimax/MiniMax-M3",
            "mistralai/magistral-medium-2509",
            "mistralai/magistral-small-2509",
            "mistralai/mistral-large-2512",
            "mistralai/mistral-medium-2505",
            "mistralai/mistral-small-2503",
            "nvidia/nemotron-3-ultra-550b-a55b",
            "openai/gpt-3.5-turbo",
            "openai/gpt-4-turbo",
            "openai/gpt-4.1-2025-04-14",
            "openai/gpt-4.1-mini-2025-04-14",
            "openai/gpt-4.1-nano-2025-04-14",
            "openai/gpt-4o-2024-08-06",
            "openai/gpt-4o-2024-11-20",
            "openai/gpt-5-2025-08-07",
            "openai/gpt-5-mini-2025-08-07",
            "openai/gpt-5-nano-2025-08-07",
            "openai/gpt-5.1-2025-11-13",
            "openai/gpt-5.2-2025-12-11",
            "openai/gpt-5.4-2026-03-05",
            "openai/gpt-5.4-nano-2026-03-17",
            "openai/gpt-5.5",
            "openai/gpt-5.6-luna",
            "openai/gpt-5.6-sol",
            "openai/gpt-5.6-terra",
            "openai/o1-2024-12-17",
            "openai/o3-2025-04-16",
            "openai/o3-mini-2025-01-31",
            "openai/o4-mini-2025-04-16",
            "poolside/laguna-m.1",
            "poolside/laguna-xs.2",
            "tencent/hy4-preview",
            "thinkingmachines/inkling",
            "thinkingmachines/inkling-small",
            "together/Qwen/Qwen2.5-72B-Instruct-Turbo",
            "together/Qwen/Qwen2.5-7B-Instruct-Turbo",
            "together/google/gemma-2-27b-it",
            "together/google/gemma-2-9b-it",
            "together/meta-llama/Llama-2-70b-hf",
            "together/meta-llama/Llama-3.3-70B-Instruct-Turbo",
            "together/meta-llama/Llama-4-Scout-17B-16E-Instruct",
            "together/mistralai/Mistral-7B-v0.1",
            "together/mistralai/Mixtral-8x7B-v0.1",
            "together/moonshotai/Kimi-K2-Instruct",
            "together/togethercomputer/llama-2-13b",
            "together/togethercomputer/llama-2-7b",
            "xiaomi/mimo-v2.5",
            "xiaomi/mimo-v2.5-pro",
            "zai/glm-4.5",
            "zai/glm-4.6",
            "zai/glm-4.7",
            "zai/glm-5-thinking",
            "zai/glm-5.1",
            "zai/glm-5.2",
            "zai/glm-5.3",
            "zai/glm-5.3-flash"
        ],
        "partners": [
            {
                "name": "codex",
                "url": "https://law.stanford.edu/codex-the-stanford-center-for-legal-informatics/"
            },
            {
                "name": "lawbeta",
                "url": "https://lawbeta.github.io/"
            },
            {
                "name": "tai",
                "url": "https://www.together.ai/"
            }
        ],
        "showBadge": false,
        "visible": true,
        "use_cost_per_test": false,
        "runner": "platform",
        "mode": "one-shot",
        "archived": false,
        "partner": false,
        "total_models": 149
    },
    "tasks": {
        "overall": {
            "anthropic/claude-fable-5": {
                "accuracy": 88.561,
                "latency": 8.964,
                "stderr": 0.331,
                "cost_per_test": 0.030639,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-fable-5-1": {
                "accuracy": 88.511,
                "latency": 12.405,
                "stderr": 0.42,
                "cost_per_test": 0.051606,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-4-argon": {
                "accuracy": 88.299,
                "latency": 12.623,
                "stderr": 0.366,
                "cost_per_test": 0.016117,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 87.398,
                "latency": 10.062,
                "stderr": 0.329,
                "cost_per_test": 0.006771,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.7-flash": {
                "accuracy": 87.256,
                "latency": 2.227,
                "stderr": 0.416,
                "cost_per_test": 0.00409,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 87.025,
                "latency": 8.327,
                "stderr": 0.368,
                "cost_per_test": 0.01046,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.8-flash": {
                "accuracy": 86.993,
                "latency": 3.32,
                "stderr": 0.432,
                "cost_per_test": 0.008109,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-5": {
                "accuracy": 86.974,
                "latency": 4.415,
                "stderr": 0.417,
                "cost_per_test": 0.011976,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.6-sol": {
                "accuracy": 86.965,
                "latency": 6.198,
                "stderr": 0.412,
                "cost_per_test": 0.012678,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 86.858,
                "latency": 4.673,
                "stderr": 0.356,
                "cost_per_test": 0.002067,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.6-flash": {
                "accuracy": 86.705,
                "latency": 2.962,
                "stderr": 0.414,
                "cost_per_test": 0.004766,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 86.515,
                "latency": 18.139,
                "stderr": 0.41,
                "cost_per_test": 0.015403,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4.6": {
                "accuracy": 86.307,
                "latency": 27.42,
                "stderr": 0.418,
                "cost_per_test": 0.01457,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "kimi/kimi-k3": {
                "accuracy": 86.212,
                "latency": 7.106,
                "stderr": 0.412,
                "cost_per_test": 0.007921,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 86.044,
                "latency": 27.787,
                "stderr": 0.414,
                "cost_per_test": 0.023053,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 86.023,
                "latency": 18.59,
                "stderr": 0.379,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4.5": {
                "accuracy": 85.965,
                "latency": 67.875,
                "stderr": 0.413,
                "cost_per_test": 0.004535,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 85.683,
                "latency": 7.048,
                "stderr": 0.397,
                "cost_per_test": 0.004796,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "minimax/MiniMax-M3": {
                "accuracy": 85.415,
                "latency": 8.205,
                "stderr": 0.436,
                "cost_per_test": 0.00091,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 512000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 85.301,
                "latency": 3.755,
                "stderr": 0.368,
                "cost_per_test": 0.004043,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "meta/muse_spark_1_2": {
                "accuracy": 85.262,
                "latency": 30.102,
                "stderr": 0.446,
                "cost_per_test": 0.00637,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "anthropic/claude-opus-4-7": {
                "accuracy": 85.251,
                "latency": 26.622,
                "stderr": 0.451,
                "cost_per_test": 0.008332,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.6-terra": {
                "accuracy": 85.105,
                "latency": 2.782,
                "stderr": 0.45,
                "cost_per_test": 0.002643,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3.5-plus-thinking": {
                "accuracy": 85.104,
                "latency": 30.422,
                "stderr": 0.408,
                "cost_per_test": 0.003671,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "meta/muse_spark_1_1": {
                "accuracy": 84.981,
                "latency": 17.148,
                "stderr": 0.425,
                "cost_per_test": 0.008535,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "alibaba/qwen3.7-max": {
                "accuracy": 84.913,
                "latency": 11.853,
                "stderr": 0.413,
                "cost_per_test": 0.005967,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "zai/glm-5.3": {
                "accuracy": 84.837,
                "latency": 22.484,
                "stderr": 0.4,
                "cost_per_test": 0.006513,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 84.738,
                "latency": 50.067,
                "stderr": 0.449,
                "cost_per_test": 0.0065,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 84.604,
                "latency": 7.638,
                "stderr": 0.391,
                "cost_per_test": 0.037097,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-4.3": {
                "accuracy": 84.458,
                "latency": 348.006,
                "stderr": 0.424,
                "cost_per_test": 0.003055,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "zai/glm-5.1": {
                "accuracy": 84.394,
                "latency": 13.749,
                "stderr": 0.429,
                "cost_per_test": 0.002183,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "grok/grok-4.7": {
                "accuracy": 84.389,
                "latency": 26.019,
                "stderr": 0.456,
                "cost_per_test": 0.010658,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-2.5-pro-exp-03-25": {
                "accuracy": 84.322,
                "latency": 4.022,
                "stderr": 0.408,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3.5-flash": {
                "accuracy": 84.276,
                "latency": 16.73,
                "stderr": 0.41,
                "cost_per_test": 0.00076,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 84.233,
                "latency": 19.073,
                "stderr": 0.422,
                "cost_per_test": 0.003042,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "meta/muse_spark": {
                "accuracy": 84.217,
                "latency": 34.203,
                "stderr": 0.416,
                "cost_per_test": 0.001178,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 84.084,
                "latency": 13.286,
                "stderr": 0.447,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.5-flash-lite": {
                "accuracy": 84.073,
                "latency": 1.666,
                "stderr": 0.437,
                "cost_per_test": 0.00128,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "zai/glm-5.2": {
                "accuracy": 84.073,
                "latency": 5.766,
                "stderr": 0.452,
                "cost_per_test": 0.001596,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "zai/glm-5-thinking": {
                "accuracy": 84.059,
                "latency": 16.056,
                "stderr": 0.411,
                "cost_per_test": 0.002311,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "openai/gpt-5.6-luna": {
                "accuracy": 84.034,
                "latency": 5.886,
                "stderr": 0.421,
                "cost_per_test": 0.000942,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 83.98,
                "latency": 8.916,
                "stderr": 0.411,
                "cost_per_test": 0.000626,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "zai/glm-5.3-flash": {
                "accuracy": 83.928,
                "latency": 12.521,
                "stderr": 0.427,
                "cost_per_test": 0.000278,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "anthropic/claude-sonnet-5": {
                "accuracy": 83.922,
                "latency": 4.901,
                "stderr": 0.458,
                "cost_per_test": 0.005759,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-04-17": {
                "accuracy": 83.796,
                "latency": 0.545,
                "stderr": 0.399,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 83.764,
                "latency": 2.704,
                "stderr": 0.406,
                "cost_per_test": 0.000295,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/o3-2025-04-16": {
                "accuracy": 83.761,
                "latency": 5.979,
                "stderr": 0.424,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "tencent/hy4-preview": {
                "accuracy": 83.757,
                "latency": 90.556,
                "stderr": 0.412,
                "cost_per_test": 0.012913,
                "temperature": 0.9,
                "top_p": 1.0,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Tencent",
                "harness": null
            },
            "alibaba/qwen3.8-max": {
                "accuracy": 83.609,
                "latency": 36.373,
                "stderr": 0.418,
                "cost_per_test": 0.010412,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "google/gemini-3.5-flash": {
                "accuracy": 83.602,
                "latency": 34.576,
                "stderr": 0.849,
                "cost_per_test": 0.007553,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-4-8": {
                "accuracy": 83.568,
                "latency": 3.106,
                "stderr": 0.474,
                "cost_per_test": 0.009106,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-4-1-20250805": {
                "accuracy": 83.458,
                "latency": 2.79,
                "stderr": 0.449,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "zai/glm-4.7": {
                "accuracy": 83.358,
                "latency": 30.675,
                "stderr": 0.473,
                "cost_per_test": 0.002348,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "deepseek/deepseek-v4.1-flash": {
                "accuracy": 83.277,
                "latency": 6.533,
                "stderr": 0.458,
                "cost_per_test": 0.001419,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "grok/grok-4-0709": {
                "accuracy": 83.192,
                "latency": 27.346,
                "stderr": 0.469,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "grok/grok-3-mini-fast-high-reasoning": {
                "accuracy": 83.143,
                "latency": 5.095,
                "stderr": 0.42,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-4.1-2025-04-14": {
                "accuracy": 83.1,
                "latency": 0.509,
                "stderr": 0.427,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-4-20250514": {
                "accuracy": 83.071,
                "latency": 3.733,
                "stderr": 0.439,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "inception/mercury-2.5": {
                "accuracy": 83.06,
                "latency": 3.908,
                "stderr": 0.482,
                "cost_per_test": 0.000787,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Inception",
                "harness": null
            },
            "thinkingmachines/inkling-small": {
                "accuracy": 82.986,
                "latency": 17.655,
                "stderr": 0.459,
                "cost_per_test": 0.001332,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 262000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514": {
                "accuracy": 82.954,
                "latency": 4.849,
                "stderr": 0.406,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "thinkingmachines/inkling": {
                "accuracy": 82.949,
                "latency": 23.142,
                "stderr": 0.496,
                "cost_per_test": 0.013327,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101": {
                "accuracy": 82.837,
                "latency": 2.376,
                "stderr": 0.434,
                "cost_per_test": 0.01319,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 82.764,
                "latency": 17.138,
                "stderr": 0.395,
                "cost_per_test": 0.017566,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025-thinking": {
                "accuracy": 82.625,
                "latency": 1.965,
                "stderr": 0.414,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-3": {
                "accuracy": 82.595,
                "latency": 0.741,
                "stderr": 0.394,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-3-7-sonnet-20250219-thinking": {
                "accuracy": 82.519,
                "latency": 6.536,
                "stderr": 0.53,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 5096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025": {
                "accuracy": 82.486,
                "latency": 1.154,
                "stderr": 0.431,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 82.451,
                "latency": 8.579,
                "stderr": 0.457,
                "cost_per_test": 0.000384,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "alibaba/qwen3.8-27b": {
                "accuracy": 82.428,
                "latency": 47.806,
                "stderr": 0.474,
                "cost_per_test": 0.004666,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "deepseek/deepseek-v4-pro-0813": {
                "accuracy": 82.361,
                "latency": 20.049,
                "stderr": 0.437,
                "cost_per_test": 0.004081,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "openai/gpt-4o-2024-11-20": {
                "accuracy": 82.214,
                "latency": 0.452,
                "stderr": 0.463,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 82.12,
                "latency": 21.628,
                "stderr": 0.476,
                "cost_per_test": 0.004202,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "nvidia/nemotron-3-ultra-550b-a55b": {
                "accuracy": 82.066,
                "latency": 5.179,
                "stderr": 0.46,
                "cost_per_test": null,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Nvidia",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514-thinking": {
                "accuracy": 82.065,
                "latency": 10.605,
                "stderr": 0.455,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025-thinking": {
                "accuracy": 82.044,
                "latency": 3.139,
                "stderr": 0.438,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-3-mini-fast-low-reasoning": {
                "accuracy": 81.925,
                "latency": 3.44,
                "stderr": 0.434,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "low",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "alibaba/qwen3-max": {
                "accuracy": 81.861,
                "latency": 0.0,
                "stderr": 0.399,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 81.77,
                "latency": 9.432,
                "stderr": 0.452,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "together/moonshotai/Kimi-K2-Instruct": {
                "accuracy": 81.454,
                "latency": 1.172,
                "stderr": 0.444,
                "cost_per_test": null,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 81.238,
                "latency": 10.372,
                "stderr": 0.487,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "fireworks/deepseek-v3": {
                "accuracy": 80.762,
                "latency": 4.395,
                "stderr": 0.429,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "grok/grok-4-fast-reasoning": {
                "accuracy": 80.601,
                "latency": 3.803,
                "stderr": 0.453,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-4-turbo": {
                "accuracy": 80.462,
                "latency": 1.405,
                "stderr": 0.482,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/o1-2024-12-17": {
                "accuracy": 80.393,
                "latency": 8.338,
                "stderr": 0.578,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 2048,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3-max-preview": {
                "accuracy": 80.333,
                "latency": 1.886,
                "stderr": 0.397,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 80.323,
                "latency": 184.46,
                "stderr": 0.468,
                "cost_per_test": 0.004788,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "kimi/kimi-k2-thinking": {
                "accuracy": 80.201,
                "latency": 43.991,
                "stderr": 0.459,
                "cost_per_test": 0.002182,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "fireworks/qwen3-235b-a22b": {
                "accuracy": 80.179,
                "latency": 12.366,
                "stderr": 0.441,
                "cost_per_test": null,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "openai/gpt-4o-2024-08-06": {
                "accuracy": 80.12,
                "latency": 0.532,
                "stderr": 0.538,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-3-7-sonnet-20250219": {
                "accuracy": 80.001,
                "latency": 1.199,
                "stderr": 0.484,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "minimax/MiniMax-M2.5": {
                "accuracy": 79.963,
                "latency": 15.819,
                "stderr": 0.473,
                "cost_per_test": 0.001287,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "cohere/command-a-03-2025": {
                "accuracy": 79.697,
                "latency": 0.797,
                "stderr": 0.422,
                "cost_per_test": null,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 8000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "ant/ling-3.0-flash-2607": {
                "accuracy": 79.69,
                "latency": 3.896,
                "stderr": 0.515,
                "cost_per_test": 0.000172,
                "temperature": 0.6,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "zai/glm-4.6": {
                "accuracy": 79.608,
                "latency": 29.819,
                "stderr": 0.522,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 122880,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "together/Qwen/Qwen2.5-72B-Instruct-Turbo": {
                "accuracy": 79.403,
                "latency": 0.831,
                "stderr": 0.66,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "openai/o4-mini-2025-04-16": {
                "accuracy": 79.185,
                "latency": 4.03,
                "stderr": 0.498,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "mistralai/mistral-large-2512": {
                "accuracy": 79.138,
                "latency": 1.731,
                "stderr": 0.46,
                "cost_per_test": 0.000439,
                "temperature": 0.065,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "grok/grok-4-1-fast-non-reasoning": {
                "accuracy": 79.113,
                "latency": 1.757,
                "stderr": 0.46,
                "cost_per_test": 9.1e-05,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025": {
                "accuracy": 79.008,
                "latency": 0.456,
                "stderr": 0.443,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "xiaomi/mimo-v2.5": {
                "accuracy": 78.886,
                "latency": 11.184,
                "stderr": 0.506,
                "cost_per_test": 0.000264,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "grok/grok-4-fast-non-reasoning": {
                "accuracy": 78.396,
                "latency": 0.933,
                "stderr": 0.472,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-2.0-flash-001": {
                "accuracy": 78.364,
                "latency": 0.447,
                "stderr": 0.782,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-4.1-mini-2025-04-14": {
                "accuracy": 78.044,
                "latency": 0.467,
                "stderr": 0.419,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 77.92,
                "latency": 2.697,
                "stderr": 0.427,
                "cost_per_test": 0.000114,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "fireworks/llama4-maverick-instruct-basic": {
                "accuracy": 77.812,
                "latency": 0.472,
                "stderr": 0.42,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 77.738,
                "latency": 7.574,
                "stderr": 0.48,
                "cost_per_test": 0.006065,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "fireworks/deepseek-v3-0324": {
                "accuracy": 77.727,
                "latency": 7.712,
                "stderr": 0.418,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "deepseek/deepseek-v4-flash-0731": {
                "accuracy": 77.714,
                "latency": 3.947,
                "stderr": 0.502,
                "cost_per_test": 0.000523,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "together/meta-llama/Llama-3.3-70B-Instruct-Turbo": {
                "accuracy": 77.18,
                "latency": 0.548,
                "stderr": 0.771,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "xiaomi/mimo-v2.5-pro": {
                "accuracy": 77.129,
                "latency": 27.142,
                "stderr": 0.55,
                "cost_per_test": 0.000845,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "fireworks/deepseek-v3p2-thinking": {
                "accuracy": 76.076,
                "latency": 31.468,
                "stderr": 0.493,
                "cost_per_test": 0.000454,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "fireworks/gpt-oss-120b": {
                "accuracy": 75.938,
                "latency": 7.852,
                "stderr": 0.517,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "zai/glm-4.5": {
                "accuracy": 75.627,
                "latency": 16.402,
                "stderr": 0.511,
                "cost_per_test": null,
                "temperature": 0.6,
                "top_p": 1.0,
                "max_output_tokens": 81920,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "minimax/MiniMax-M2.1": {
                "accuracy": 75.448,
                "latency": 11.491,
                "stderr": 0.539,
                "cost_per_test": 0.000937,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "poolside/laguna-m.1": {
                "accuracy": 75.144,
                "latency": 42.363,
                "stderr": 0.591,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "ai21labs/jamba-1.5-large": {
                "accuracy": 74.16,
                "latency": 0.808,
                "stderr": 0.639,
                "cost_per_test": null,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "together/meta-llama/Llama-4-Scout-17B-16E-Instruct": {
                "accuracy": 72.036,
                "latency": 0.392,
                "stderr": 0.479,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "ai21labs/jamba-large-1.6": {
                "accuracy": 71.962,
                "latency": 0.814,
                "stderr": 0.424,
                "cost_per_test": null,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "openai/o3-mini-2025-01-31": {
                "accuracy": 71.539,
                "latency": 12.011,
                "stderr": 0.72,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "poolside/laguna-xs.2": {
                "accuracy": 71.026,
                "latency": 11.924,
                "stderr": 0.563,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "fireworks/gpt-oss-20b": {
                "accuracy": 70.849,
                "latency": 7.187,
                "stderr": 0.61,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "anthropic/claude-3-5-haiku-20241022": {
                "accuracy": 70.331,
                "latency": 1.066,
                "stderr": 0.64,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-1.0-pro-002": {
                "accuracy": 70.247,
                "latency": 0.465,
                "stderr": 0.765,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "ai21labs/jamba-mini-1.6": {
                "accuracy": 69.726,
                "latency": 0.305,
                "stderr": 0.515,
                "cost_per_test": null,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "together/Qwen/Qwen2.5-7B-Instruct-Turbo": {
                "accuracy": 69.559,
                "latency": 0.395,
                "stderr": 0.783,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "mistralai/mistral-small-2503": {
                "accuracy": 69.157,
                "latency": 0.583,
                "stderr": 0.512,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "google/gemini-1.5-pro-002": {
                "accuracy": 69.08,
                "latency": 0.898,
                "stderr": 0.751,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "ant/ling-3.0-flash-af-rc3": {
                "accuracy": 68.743,
                "latency": 6.35,
                "stderr": 0.525,
                "cost_per_test": 0.000201,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "cohere/command-r-plus": {
                "accuracy": 68.287,
                "latency": 0.396,
                "stderr": 0.65,
                "cost_per_test": null,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 4000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "together/google/gemma-2-9b-it": {
                "accuracy": 68.138,
                "latency": 0.579,
                "stderr": 0.735,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "together/google/gemma-2-27b-it": {
                "accuracy": 67.978,
                "latency": 0.887,
                "stderr": 0.651,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "fireworks/deepseek-r1": {
                "accuracy": 67.323,
                "latency": 18.549,
                "stderr": 0.535,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 2048,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "ai21labs/jamba-1.5-mini": {
                "accuracy": 66.616,
                "latency": 0.275,
                "stderr": 0.673,
                "cost_per_test": null,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "fireworks/deepseek-v3p2": {
                "accuracy": 65.962,
                "latency": 2.185,
                "stderr": 0.419,
                "cost_per_test": 8.6e-05,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "openai/gpt-3.5-turbo": {
                "accuracy": 64.372,
                "latency": 0.542,
                "stderr": 0.628,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-1.5-flash-002": {
                "accuracy": 63.238,
                "latency": 0.476,
                "stderr": 0.736,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "together/meta-llama/Llama-2-70b-hf": {
                "accuracy": 62.434,
                "latency": 14.204,
                "stderr": 0.711,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "mistralai/mistral-medium-2505": {
                "accuracy": 61.979,
                "latency": 0.676,
                "stderr": 0.473,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "cohere/command-a-plus-05-2026": {
                "accuracy": 61.422,
                "latency": 22.648,
                "stderr": 0.627,
                "cost_per_test": 0.033012,
                "temperature": 0.9,
                "top_p": 0.95,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "openai/gpt-4.1-nano-2025-04-14": {
                "accuracy": 61.056,
                "latency": 0.357,
                "stderr": 0.595,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "together/mistralai/Mixtral-8x7B-v0.1": {
                "accuracy": 55.836,
                "latency": 14.142,
                "stderr": 0.742,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "mistralai/magistral-medium-2509": {
                "accuracy": 54.767,
                "latency": 11.734,
                "stderr": 0.561,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "together/mistralai/Mistral-7B-v0.1": {
                "accuracy": 53.77,
                "latency": 1.715,
                "stderr": 0.779,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "together/togethercomputer/llama-2-13b": {
                "accuracy": 53.696,
                "latency": 1.673,
                "stderr": 0.733,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "together/togethercomputer/llama-2-7b": {
                "accuracy": 51.869,
                "latency": 0.984,
                "stderr": 0.63,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "openai/gpt-5-nano-2025-08-07": {
                "accuracy": 50.129,
                "latency": 22.496,
                "stderr": 0.472,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "mistralai/magistral-small-2509": {
                "accuracy": 40.023,
                "latency": 6.67,
                "stderr": 0.472,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "cohere/command-r": {
                "accuracy": 32.965,
                "latency": 0.244,
                "stderr": 0.71,
                "cost_per_test": null,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 4000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            }
        },
        "issue_tasks": {
            "google/gemini-4-argon": {
                "accuracy": 92.941,
                "latency": 8.441,
                "stderr": 0.497,
                "cost_per_test": 0.014298,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-fable-5-1": {
                "accuracy": 92.72,
                "latency": 6.13,
                "stderr": 0.498,
                "cost_per_test": 0.037363,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.7-flash": {
                "accuracy": 92.146,
                "latency": 2.017,
                "stderr": 0.508,
                "cost_per_test": 0.004487,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-fable-5": {
                "accuracy": 92.064,
                "latency": 6.922,
                "stderr": 0.505,
                "cost_per_test": 0.030005,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-5": {
                "accuracy": 91.878,
                "latency": 2.886,
                "stderr": 0.527,
                "cost_per_test": 0.015445,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 91.48,
                "latency": 8.723,
                "stderr": 0.53,
                "cost_per_test": 0.006807,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.6-sol": {
                "accuracy": 91.438,
                "latency": 3.001,
                "stderr": 0.535,
                "cost_per_test": 0.008536,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3.8-flash": {
                "accuracy": 91.278,
                "latency": 2.361,
                "stderr": 0.533,
                "cost_per_test": 0.006227,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 91.193,
                "latency": 6.799,
                "stderr": 0.547,
                "cost_per_test": 0.00728,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 91.172,
                "latency": 7.218,
                "stderr": 0.548,
                "cost_per_test": 0.010184,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3.5-plus-thinking": {
                "accuracy": 91.166,
                "latency": 26.931,
                "stderr": 0.538,
                "cost_per_test": 0.003432,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 91.145,
                "latency": 4.207,
                "stderr": 0.537,
                "cost_per_test": 0.002102,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.6-flash": {
                "accuracy": 91.121,
                "latency": 2.519,
                "stderr": 0.557,
                "cost_per_test": 0.005112,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "kimi/kimi-k3": {
                "accuracy": 90.911,
                "latency": 5.433,
                "stderr": 0.528,
                "cost_per_test": 0.004174,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "alibaba/qwen3.5-flash": {
                "accuracy": 90.869,
                "latency": 17.011,
                "stderr": 0.547,
                "cost_per_test": 0.000824,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "minimax/MiniMax-M3": {
                "accuracy": 90.862,
                "latency": 6.976,
                "stderr": 0.553,
                "cost_per_test": 0.0009,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 512000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 90.685,
                "latency": 3.921,
                "stderr": 0.542,
                "cost_per_test": 0.006478,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3.8-max": {
                "accuracy": 90.635,
                "latency": 8.901,
                "stderr": 0.556,
                "cost_per_test": 0.003537,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 90.526,
                "latency": 9.338,
                "stderr": 0.569,
                "cost_per_test": 0.008692,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-5.3": {
                "accuracy": 90.412,
                "latency": 5.117,
                "stderr": 0.544,
                "cost_per_test": 0.002907,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "alibaba/qwen3.8-27b": {
                "accuracy": 90.175,
                "latency": 17.689,
                "stderr": 0.555,
                "cost_per_test": 0.00184,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "zai/glm-5.3-flash": {
                "accuracy": 90.138,
                "latency": 5.181,
                "stderr": 0.56,
                "cost_per_test": 0.000142,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "grok/grok-4.6": {
                "accuracy": 90.129,
                "latency": 23.504,
                "stderr": 0.582,
                "cost_per_test": 0.013901,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "grok/grok-4.5": {
                "accuracy": 90.129,
                "latency": 85.022,
                "stderr": 0.573,
                "cost_per_test": 0.004414,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 90.089,
                "latency": 24.471,
                "stderr": 0.559,
                "cost_per_test": 0.003533,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "tencent/hy4-preview": {
                "accuracy": 90.029,
                "latency": 51.782,
                "stderr": 0.575,
                "cost_per_test": 0.007285,
                "temperature": 0.9,
                "top_p": 1.0,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Tencent",
                "harness": null
            },
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 90.011,
                "latency": 5.379,
                "stderr": 0.566,
                "cost_per_test": 0.004186,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3.7-max": {
                "accuracy": 89.731,
                "latency": 7.485,
                "stderr": 0.571,
                "cost_per_test": 0.005988,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 89.727,
                "latency": 13.911,
                "stderr": 0.567,
                "cost_per_test": 0.002679,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 89.55,
                "latency": 10.666,
                "stderr": 0.581,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-5.2": {
                "accuracy": 89.506,
                "latency": 2.862,
                "stderr": 0.575,
                "cost_per_test": 0.001761,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "grok/grok-4.7": {
                "accuracy": 89.495,
                "latency": 22.095,
                "stderr": 0.577,
                "cost_per_test": 0.009779,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-04-17": {
                "accuracy": 89.456,
                "latency": 0.425,
                "stderr": 0.558,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-4-0709": {
                "accuracy": 89.425,
                "latency": 15.489,
                "stderr": 0.553,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-5.6-terra": {
                "accuracy": 89.392,
                "latency": 2.159,
                "stderr": 0.567,
                "cost_per_test": 0.002836,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "meta/muse_spark_1_2": {
                "accuracy": 89.374,
                "latency": 21.721,
                "stderr": 0.58,
                "cost_per_test": 0.005294,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 89.371,
                "latency": 2.392,
                "stderr": 0.57,
                "cost_per_test": 0.000329,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.5-flash-lite": {
                "accuracy": 89.325,
                "latency": 1.33,
                "stderr": 0.582,
                "cost_per_test": 0.001185,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 89.256,
                "latency": 7.239,
                "stderr": 0.583,
                "cost_per_test": 0.000422,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "grok/grok-4.3": {
                "accuracy": 89.182,
                "latency": 275.969,
                "stderr": 0.57,
                "cost_per_test": 0.003143,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "inception/mercury-2.5": {
                "accuracy": 89.051,
                "latency": 3.586,
                "stderr": 0.569,
                "cost_per_test": 0.000911,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Inception",
                "harness": null
            },
            "zai/glm-5.1": {
                "accuracy": 89.02,
                "latency": 10.961,
                "stderr": 0.588,
                "cost_per_test": 0.001949,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "meta/muse_spark_1_1": {
                "accuracy": 88.993,
                "latency": 16.813,
                "stderr": 0.571,
                "cost_per_test": 0.009923,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "anthropic/claude-opus-4-7": {
                "accuracy": 88.947,
                "latency": 27.488,
                "stderr": 0.573,
                "cost_per_test": 0.013182,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 88.866,
                "latency": 8.434,
                "stderr": 0.592,
                "cost_per_test": 0.04999,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "thinkingmachines/inkling-small": {
                "accuracy": 88.854,
                "latency": 9.708,
                "stderr": 0.595,
                "cost_per_test": 0.000826,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 262000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "meta/muse_spark": {
                "accuracy": 88.799,
                "latency": 38.275,
                "stderr": 0.577,
                "cost_per_test": 0.001456,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "deepseek/deepseek-v4.1-flash": {
                "accuracy": 88.753,
                "latency": 4.229,
                "stderr": 0.597,
                "cost_per_test": 0.000887,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "zai/glm-4.7": {
                "accuracy": 88.746,
                "latency": 26.198,
                "stderr": 0.595,
                "cost_per_test": 0.002019,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514-thinking": {
                "accuracy": 88.693,
                "latency": 9.827,
                "stderr": 0.555,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.6-luna": {
                "accuracy": 88.68,
                "latency": 2.819,
                "stderr": 0.591,
                "cost_per_test": 0.000489,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 88.569,
                "latency": 12.52,
                "stderr": 0.584,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-4-1-20250805": {
                "accuracy": 88.534,
                "latency": 2.692,
                "stderr": 0.579,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 88.305,
                "latency": 6.548,
                "stderr": 0.591,
                "cost_per_test": 0.007325,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-sonnet-5": {
                "accuracy": 88.24,
                "latency": 2.196,
                "stderr": 0.609,
                "cost_per_test": 0.00574,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-3-mini-fast-high-reasoning": {
                "accuracy": 88.201,
                "latency": 4.359,
                "stderr": 0.563,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-opus-4-20250514": {
                "accuracy": 87.936,
                "latency": 3.775,
                "stderr": 0.586,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-2.0-flash-001": {
                "accuracy": 87.905,
                "latency": 0.334,
                "stderr": 0.585,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-4-8": {
                "accuracy": 87.834,
                "latency": 2.135,
                "stderr": 0.601,
                "cost_per_test": 0.013321,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-3": {
                "accuracy": 87.812,
                "latency": 0.386,
                "stderr": 0.566,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "zai/glm-5-thinking": {
                "accuracy": 87.755,
                "latency": 8.909,
                "stderr": 0.613,
                "cost_per_test": 0.00167,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "together/moonshotai/Kimi-K2-Instruct": {
                "accuracy": 87.675,
                "latency": 0.881,
                "stderr": 0.58,
                "cost_per_test": null,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "grok/grok-4-fast-reasoning": {
                "accuracy": 87.657,
                "latency": 4.673,
                "stderr": 0.593,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/o3-2025-04-16": {
                "accuracy": 87.587,
                "latency": 4.6,
                "stderr": 0.574,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-4o-2024-11-20": {
                "accuracy": 87.568,
                "latency": 0.381,
                "stderr": 0.615,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025-thinking": {
                "accuracy": 87.565,
                "latency": 1.881,
                "stderr": 0.569,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "nvidia/nemotron-3-ultra-550b-a55b": {
                "accuracy": 87.318,
                "latency": 2.071,
                "stderr": 0.59,
                "cost_per_test": null,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Nvidia",
                "harness": null
            },
            "grok/grok-3-mini-fast-low-reasoning": {
                "accuracy": 87.28,
                "latency": 3.289,
                "stderr": 0.542,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "low",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "thinkingmachines/inkling": {
                "accuracy": 87.162,
                "latency": 12.791,
                "stderr": 0.606,
                "cost_per_test": 0.008901,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "openai/gpt-4.1-2025-04-14": {
                "accuracy": 87.158,
                "latency": 0.418,
                "stderr": 0.592,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 87.118,
                "latency": 8.837,
                "stderr": 0.64,
                "cost_per_test": 0.000927,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "google/gemini-2.5-pro-exp-03-25": {
                "accuracy": 87.099,
                "latency": 2.709,
                "stderr": 0.585,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 87.09,
                "latency": 6.488,
                "stderr": 0.574,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 86.743,
                "latency": 12.412,
                "stderr": 0.634,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-3-7-sonnet-20250219-thinking": {
                "accuracy": 86.619,
                "latency": 6.353,
                "stderr": 0.588,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 5096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 86.574,
                "latency": 20.233,
                "stderr": 0.593,
                "cost_per_test": 0.005224,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025": {
                "accuracy": 86.447,
                "latency": 0.402,
                "stderr": 0.619,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "xiaomi/mimo-v2.5": {
                "accuracy": 86.337,
                "latency": 12.928,
                "stderr": 0.626,
                "cost_per_test": 0.000331,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025": {
                "accuracy": 86.31,
                "latency": 1.015,
                "stderr": 0.609,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "fireworks/deepseek-v3-0324": {
                "accuracy": 86.204,
                "latency": 7.862,
                "stderr": 0.609,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "fireworks/qwen3-235b-a22b": {
                "accuracy": 86.016,
                "latency": 8.841,
                "stderr": 0.645,
                "cost_per_test": null,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "deepseek/deepseek-v4-pro-0813": {
                "accuracy": 85.989,
                "latency": 8.552,
                "stderr": 0.661,
                "cost_per_test": 0.002262,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "ant/ling-3.0-flash-2607": {
                "accuracy": 85.929,
                "latency": 2.805,
                "stderr": 0.633,
                "cost_per_test": 0.000121,
                "temperature": 0.6,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "grok/grok-4-1-fast-non-reasoning": {
                "accuracy": 85.753,
                "latency": 3.897,
                "stderr": 0.643,
                "cost_per_test": 0.000159,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "fireworks/deepseek-v3": {
                "accuracy": 85.619,
                "latency": 2.697,
                "stderr": 0.608,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025-thinking": {
                "accuracy": 85.416,
                "latency": 2.956,
                "stderr": 0.613,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514": {
                "accuracy": 85.113,
                "latency": 4.891,
                "stderr": 0.6,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "cohere/command-a-03-2025": {
                "accuracy": 84.798,
                "latency": 0.762,
                "stderr": 0.615,
                "cost_per_test": null,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 8000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 84.587,
                "latency": 283.655,
                "stderr": 0.671,
                "cost_per_test": 0.00519,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "alibaba/qwen3-max": {
                "accuracy": 84.574,
                "latency": 0.0,
                "stderr": 0.607,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "zai/glm-4.6": {
                "accuracy": 84.529,
                "latency": 16.281,
                "stderr": 0.681,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 122880,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "minimax/MiniMax-M2.5": {
                "accuracy": 84.169,
                "latency": 14.901,
                "stderr": 0.662,
                "cost_per_test": 0.001305,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101": {
                "accuracy": 84.152,
                "latency": 2.225,
                "stderr": 0.638,
                "cost_per_test": 0.024892,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "ant/ling-3.0-flash-af-rc3": {
                "accuracy": 83.975,
                "latency": 3.492,
                "stderr": 0.646,
                "cost_per_test": 0.000178,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "openai/o4-mini-2025-04-16": {
                "accuracy": 83.542,
                "latency": 3.993,
                "stderr": 0.7,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "together/meta-llama/Llama-3.3-70B-Instruct-Turbo": {
                "accuracy": 83.206,
                "latency": 0.546,
                "stderr": 0.591,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "openai/o1-2024-12-17": {
                "accuracy": 83.004,
                "latency": 9.267,
                "stderr": 0.656,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 2048,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-4-turbo": {
                "accuracy": 82.88,
                "latency": 1.166,
                "stderr": 0.499,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 82.807,
                "latency": 2.274,
                "stderr": 0.654,
                "cost_per_test": 0.000139,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "together/Qwen/Qwen2.5-72B-Instruct-Turbo": {
                "accuracy": 82.193,
                "latency": 0.638,
                "stderr": 0.678,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "deepseek/deepseek-v4-flash-0731": {
                "accuracy": 81.956,
                "latency": 2.829,
                "stderr": 0.727,
                "cost_per_test": 0.000484,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "grok/grok-4-fast-non-reasoning": {
                "accuracy": 81.671,
                "latency": 1.002,
                "stderr": 0.693,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "kimi/kimi-k2-thinking": {
                "accuracy": 81.46,
                "latency": 28.773,
                "stderr": 0.628,
                "cost_per_test": 0.002043,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "mistralai/mistral-large-2512": {
                "accuracy": 81.006,
                "latency": 0.938,
                "stderr": 0.694,
                "cost_per_test": 0.0008,
                "temperature": 0.065,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "openai/gpt-4o-2024-08-06": {
                "accuracy": 80.596,
                "latency": 0.42,
                "stderr": 0.653,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "xiaomi/mimo-v2.5-pro": {
                "accuracy": 80.173,
                "latency": 32.35,
                "stderr": 0.769,
                "cost_per_test": 0.00105,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "fireworks/deepseek-r1": {
                "accuracy": 80.163,
                "latency": 12.795,
                "stderr": 0.666,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 2048,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "alibaba/qwen3-max-preview": {
                "accuracy": 79.509,
                "latency": 1.558,
                "stderr": 0.662,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/gpt-4.1-mini-2025-04-14": {
                "accuracy": 79.242,
                "latency": 0.41,
                "stderr": 0.663,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "poolside/laguna-m.1": {
                "accuracy": 78.459,
                "latency": 29.012,
                "stderr": 0.791,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "anthropic/claude-3-7-sonnet-20250219": {
                "accuracy": 77.521,
                "latency": 1.198,
                "stderr": 0.656,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "poolside/laguna-xs.2": {
                "accuracy": 76.811,
                "latency": 9.388,
                "stderr": 0.771,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "fireworks/gpt-oss-120b": {
                "accuracy": 76.602,
                "latency": 5.366,
                "stderr": 0.785,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "fireworks/llama4-maverick-instruct-basic": {
                "accuracy": 76.518,
                "latency": 1.06,
                "stderr": 0.635,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "mistralai/mistral-small-2503": {
                "accuracy": 76.453,
                "latency": 0.329,
                "stderr": 0.775,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "cohere/command-r-plus": {
                "accuracy": 76.319,
                "latency": 0.318,
                "stderr": 0.708,
                "cost_per_test": null,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 4000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "fireworks/deepseek-v3p2-thinking": {
                "accuracy": 76.035,
                "latency": 33.565,
                "stderr": 0.808,
                "cost_per_test": 0.000512,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "google/gemini-3.5-flash": {
                "accuracy": 76.014,
                "latency": 50.772,
                "stderr": 3.493,
                "cost_per_test": 0.007857,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "fireworks/gpt-oss-20b": {
                "accuracy": 74.168,
                "latency": 6.529,
                "stderr": 0.84,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "ai21labs/jamba-1.5-large": {
                "accuracy": 72.783,
                "latency": 1.011,
                "stderr": 0.77,
                "cost_per_test": null,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "google/gemini-1.0-pro-002": {
                "accuracy": 72.707,
                "latency": 0.622,
                "stderr": 0.677,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "together/Qwen/Qwen2.5-7B-Instruct-Turbo": {
                "accuracy": 72.583,
                "latency": 0.324,
                "stderr": 0.822,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "ai21labs/jamba-mini-1.6": {
                "accuracy": 71.986,
                "latency": 0.279,
                "stderr": 0.81,
                "cost_per_test": null,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "minimax/MiniMax-M2.1": {
                "accuracy": 71.973,
                "latency": 10.116,
                "stderr": 0.807,
                "cost_per_test": 0.001045,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 70.469,
                "latency": 7.387,
                "stderr": 0.872,
                "cost_per_test": 0.006328,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "together/meta-llama/Llama-4-Scout-17B-16E-Instruct": {
                "accuracy": 66.836,
                "latency": 0.234,
                "stderr": 0.727,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "openai/o3-mini-2025-01-31": {
                "accuracy": 65.64,
                "latency": 14.422,
                "stderr": 0.812,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-4.5": {
                "accuracy": 64.714,
                "latency": 4.049,
                "stderr": 0.901,
                "cost_per_test": null,
                "temperature": 0.6,
                "top_p": 1.0,
                "max_output_tokens": 81920,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "ai21labs/jamba-large-1.6": {
                "accuracy": 64.335,
                "latency": 0.851,
                "stderr": 0.773,
                "cost_per_test": null,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "google/gemini-1.5-pro-002": {
                "accuracy": 64.126,
                "latency": 0.673,
                "stderr": 0.589,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "together/google/gemma-2-9b-it": {
                "accuracy": 62.036,
                "latency": 0.461,
                "stderr": 0.779,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "ai21labs/jamba-1.5-mini": {
                "accuracy": 61.185,
                "latency": 0.279,
                "stderr": 0.912,
                "cost_per_test": null,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "anthropic/claude-3-5-haiku-20241022": {
                "accuracy": 58.652,
                "latency": 0.999,
                "stderr": 0.593,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-3.5-turbo": {
                "accuracy": 57.054,
                "latency": 0.665,
                "stderr": 0.515,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "cohere/command-a-plus-05-2026": {
                "accuracy": 55.176,
                "latency": 5.359,
                "stderr": 0.865,
                "cost_per_test": 0.008963,
                "temperature": 0.9,
                "top_p": 0.95,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "together/google/gemma-2-27b-it": {
                "accuracy": 54.941,
                "latency": 0.505,
                "stderr": 0.646,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "fireworks/deepseek-v3p2": {
                "accuracy": 54.728,
                "latency": 1.507,
                "stderr": 0.396,
                "cost_per_test": 0.000151,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "together/meta-llama/Llama-2-70b-hf": {
                "accuracy": 53.892,
                "latency": 29.839,
                "stderr": 0.355,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "together/mistralai/Mixtral-8x7B-v0.1": {
                "accuracy": 53.75,
                "latency": 28.063,
                "stderr": 0.338,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "openai/gpt-5-nano-2025-08-07": {
                "accuracy": 52.874,
                "latency": 17.937,
                "stderr": 0.956,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-4.1-nano-2025-04-14": {
                "accuracy": 52.269,
                "latency": 0.366,
                "stderr": 0.83,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-1.5-flash-002": {
                "accuracy": 51.973,
                "latency": 0.445,
                "stderr": 0.263,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "together/mistralai/Mistral-7B-v0.1": {
                "accuracy": 51.677,
                "latency": 3.546,
                "stderr": 0.176,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "together/togethercomputer/llama-2-7b": {
                "accuracy": 50.296,
                "latency": 1.792,
                "stderr": 0.262,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "together/togethercomputer/llama-2-13b": {
                "accuracy": 50.258,
                "latency": 3.318,
                "stderr": 0.163,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "mistralai/magistral-medium-2509": {
                "accuracy": 40.342,
                "latency": 12.951,
                "stderr": 0.895,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "mistralai/mistral-medium-2505": {
                "accuracy": 36.119,
                "latency": 0.47,
                "stderr": 0.731,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "mistralai/magistral-small-2509": {
                "accuracy": 19.942,
                "latency": 6.31,
                "stderr": 0.681,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "cohere/command-r": {
                "accuracy": 8.888,
                "latency": 0.22,
                "stderr": 0.357,
                "cost_per_test": null,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 4000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            }
        },
        "rule_tasks": {
            "google/gemini-4-argon": {
                "accuracy": 90.774,
                "latency": 19.214,
                "stderr": 1.123,
                "cost_per_test": 0.021453,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-fable-5": {
                "accuracy": 88.33,
                "latency": 15.686,
                "stderr": 1.145,
                "cost_per_test": 0.053962,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 88.294,
                "latency": 13.146,
                "stderr": 1.152,
                "cost_per_test": 0.00883,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-fable-5-1": {
                "accuracy": 88.125,
                "latency": 33.428,
                "stderr": 1.413,
                "cost_per_test": 0.130251,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.5-flash": {
                "accuracy": 87.996,
                "latency": 35.426,
                "stderr": 1.731,
                "cost_per_test": 0.01073,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.8-flash": {
                "accuracy": 87.076,
                "latency": 3.736,
                "stderr": 1.242,
                "cost_per_test": 0.007846,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.6-sol": {
                "accuracy": 86.443,
                "latency": 14.835,
                "stderr": 1.335,
                "cost_per_test": 0.025965,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3.7-flash": {
                "accuracy": 85.851,
                "latency": 3.067,
                "stderr": 1.359,
                "cost_per_test": 0.005009,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.6-flash": {
                "accuracy": 85.726,
                "latency": 4.165,
                "stderr": 1.488,
                "cost_per_test": 0.005592,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 85.682,
                "latency": 33.436,
                "stderr": 1.456,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 85.453,
                "latency": 50.029,
                "stderr": 1.488,
                "cost_per_test": 0.040762,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 85.291,
                "latency": 5.224,
                "stderr": 1.281,
                "cost_per_test": 0.002112,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 85.246,
                "latency": 9.824,
                "stderr": 1.415,
                "cost_per_test": 0.01197,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-4-7": {
                "accuracy": 84.996,
                "latency": 46.694,
                "stderr": 1.544,
                "cost_per_test": 0.006751,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-5": {
                "accuracy": 84.805,
                "latency": 11.197,
                "stderr": 1.477,
                "cost_per_test": 0.020142,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "meta/muse_spark_1_2": {
                "accuracy": 84.725,
                "latency": 42.418,
                "stderr": 1.597,
                "cost_per_test": 0.009008,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "grok/grok-4.6": {
                "accuracy": 84.462,
                "latency": 36.546,
                "stderr": 1.609,
                "cost_per_test": 0.018209,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "grok/grok-4.5": {
                "accuracy": 84.168,
                "latency": 67.788,
                "stderr": 1.674,
                "cost_per_test": 0.005398,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "kimi/kimi-k3": {
                "accuracy": 84.05,
                "latency": 13.902,
                "stderr": 1.498,
                "cost_per_test": 0.023231,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 84.008,
                "latency": 74.892,
                "stderr": 1.306,
                "cost_per_test": 0.059881,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 83.616,
                "latency": 10.822,
                "stderr": 1.521,
                "cost_per_test": 0.006998,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "meta/muse_spark_1_1": {
                "accuracy": 83.562,
                "latency": 22.919,
                "stderr": 1.509,
                "cost_per_test": 0.009646,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 82.377,
                "latency": 5.414,
                "stderr": 1.322,
                "cost_per_test": 0.0036,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 82.168,
                "latency": 3.143,
                "stderr": 1.576,
                "cost_per_test": 0.000287,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-3-7-sonnet-20250219": {
                "accuracy": 82.128,
                "latency": 1.495,
                "stderr": 1.896,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-3-7-sonnet-20250219-thinking": {
                "accuracy": 81.762,
                "latency": 7.506,
                "stderr": 2.107,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 5096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.5-flash-lite": {
                "accuracy": 81.708,
                "latency": 2.344,
                "stderr": 1.602,
                "cost_per_test": 0.001616,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "zai/glm-5.3": {
                "accuracy": 81.662,
                "latency": 73.336,
                "stderr": 1.612,
                "cost_per_test": 0.018427,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 81.58,
                "latency": 8.387,
                "stderr": 1.441,
                "cost_per_test": 0.030796,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-2.5-pro-exp-03-25": {
                "accuracy": 81.37,
                "latency": 6.92,
                "stderr": 1.544,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3.5-plus-thinking": {
                "accuracy": 81.325,
                "latency": 32.28,
                "stderr": 1.571,
                "cost_per_test": 0.003962,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "zai/glm-5-thinking": {
                "accuracy": 81.228,
                "latency": 25.269,
                "stderr": 1.558,
                "cost_per_test": 0.00329,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "alibaba/qwen3.7-max": {
                "accuracy": 81.154,
                "latency": 21.957,
                "stderr": 1.646,
                "cost_per_test": 0.008665,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514": {
                "accuracy": 81.134,
                "latency": 5.637,
                "stderr": 1.515,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-4.7": {
                "accuracy": 81.049,
                "latency": 41.291,
                "stderr": 1.766,
                "cost_per_test": 0.015994,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 80.988,
                "latency": 11.261,
                "stderr": 1.559,
                "cost_per_test": 0.000568,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "minimax/MiniMax-M3": {
                "accuracy": 80.972,
                "latency": 11.832,
                "stderr": 1.76,
                "cost_per_test": 0.001039,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 512000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "grok/grok-4.3": {
                "accuracy": 80.785,
                "latency": 461.433,
                "stderr": 1.674,
                "cost_per_test": 0.003536,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 80.592,
                "latency": 109.453,
                "stderr": 1.635,
                "cost_per_test": 0.01431,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "zai/glm-5.1": {
                "accuracy": 80.56,
                "latency": 19.27,
                "stderr": 1.7,
                "cost_per_test": 0.003438,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "openai/gpt-5.6-terra": {
                "accuracy": 80.528,
                "latency": 5.02,
                "stderr": 1.635,
                "cost_per_test": 0.004017,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101": {
                "accuracy": 80.32,
                "latency": 3.18,
                "stderr": 1.433,
                "cost_per_test": 0.007788,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-4-turbo": {
                "accuracy": 80.26,
                "latency": 3.282,
                "stderr": 1.897,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-4.1-2025-04-14": {
                "accuracy": 80.218,
                "latency": 0.931,
                "stderr": 1.645,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/o3-2025-04-16": {
                "accuracy": 80.187,
                "latency": 10.852,
                "stderr": 1.654,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 80.076,
                "latency": 15.124,
                "stderr": 1.637,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-4-8": {
                "accuracy": 79.976,
                "latency": 4.075,
                "stderr": 1.72,
                "cost_per_test": 0.006584,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "meta/muse_spark": {
                "accuracy": 79.9,
                "latency": 43.56,
                "stderr": 1.582,
                "cost_per_test": 0.001334,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "nvidia/nemotron-3-ultra-550b-a55b": {
                "accuracy": 79.393,
                "latency": 7.98,
                "stderr": 1.763,
                "cost_per_test": null,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Nvidia",
                "harness": null
            },
            "zai/glm-5.3-flash": {
                "accuracy": 79.373,
                "latency": 37.057,
                "stderr": 1.662,
                "cost_per_test": 0.000772,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "alibaba/qwen3-max": {
                "accuracy": 79.362,
                "latency": 0.0,
                "stderr": 1.493,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "anthropic/claude-opus-4-1-20250805": {
                "accuracy": 79.36,
                "latency": 3.763,
                "stderr": 1.619,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-sonnet-5": {
                "accuracy": 79.331,
                "latency": 7.866,
                "stderr": 1.742,
                "cost_per_test": 0.006804,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 79.223,
                "latency": 27.594,
                "stderr": 1.694,
                "cost_per_test": 0.00394,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "zai/glm-4.7": {
                "accuracy": 79.181,
                "latency": 46.168,
                "stderr": 1.635,
                "cost_per_test": 0.003212,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "mistralai/mistral-large-2512": {
                "accuracy": 79.13,
                "latency": 5.148,
                "stderr": 1.496,
                "cost_per_test": 0.000344,
                "temperature": 0.065,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "deepseek/deepseek-v4-pro-0813": {
                "accuracy": 78.978,
                "latency": 51.492,
                "stderr": 1.556,
                "cost_per_test": 0.009512,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "anthropic/claude-opus-4-20250514": {
                "accuracy": 78.92,
                "latency": 4.643,
                "stderr": 1.552,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/o1-2024-12-17": {
                "accuracy": 78.864,
                "latency": 8.742,
                "stderr": 2.414,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 2048,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-5.2": {
                "accuracy": 78.84,
                "latency": 14.626,
                "stderr": 1.795,
                "cost_per_test": 0.003114,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-04-17": {
                "accuracy": 78.796,
                "latency": 1.015,
                "stderr": 1.478,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3.5-flash": {
                "accuracy": 78.722,
                "latency": 19.626,
                "stderr": 1.557,
                "cost_per_test": 0.00085,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "thinkingmachines/inkling-small": {
                "accuracy": 78.295,
                "latency": 12.741,
                "stderr": 1.631,
                "cost_per_test": 0.000934,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 262000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025": {
                "accuracy": 78.183,
                "latency": 1.732,
                "stderr": 1.664,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "deepseek/deepseek-v4.1-flash": {
                "accuracy": 77.981,
                "latency": 14.284,
                "stderr": 1.691,
                "cost_per_test": 0.003209,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025-thinking": {
                "accuracy": 77.888,
                "latency": 2.739,
                "stderr": 1.596,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3-max-preview": {
                "accuracy": 77.87,
                "latency": 3.31,
                "stderr": 1.425,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/gpt-5.6-luna": {
                "accuracy": 77.736,
                "latency": 15.668,
                "stderr": 1.596,
                "cost_per_test": 0.002719,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-4o-2024-11-20": {
                "accuracy": 77.429,
                "latency": 0.882,
                "stderr": 1.9,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-3-mini-fast-high-reasoning": {
                "accuracy": 77.429,
                "latency": 5.591,
                "stderr": 1.567,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 77.425,
                "latency": 70.613,
                "stderr": 1.699,
                "cost_per_test": 0.005923,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "openai/o4-mini-2025-04-16": {
                "accuracy": 77.285,
                "latency": 5.51,
                "stderr": 1.676,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 77.184,
                "latency": 9.042,
                "stderr": 1.726,
                "cost_per_test": 0.006934,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025-thinking": {
                "accuracy": 76.852,
                "latency": 3.212,
                "stderr": 1.686,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "thinkingmachines/inkling": {
                "accuracy": 76.703,
                "latency": 41.232,
                "stderr": 1.877,
                "cost_per_test": 0.022792,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "grok/grok-3": {
                "accuracy": 76.659,
                "latency": 2.127,
                "stderr": 1.371,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "kimi/kimi-k2-thinking": {
                "accuracy": 76.355,
                "latency": 62.773,
                "stderr": 1.666,
                "cost_per_test": 0.002632,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "tencent/hy4-preview": {
                "accuracy": 76.302,
                "latency": 191.92,
                "stderr": 1.504,
                "cost_per_test": 0.026272,
                "temperature": 0.9,
                "top_p": 1.0,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Tencent",
                "harness": null
            },
            "deepseek/deepseek-v4-flash-0731": {
                "accuracy": 76.287,
                "latency": 6.867,
                "stderr": 1.728,
                "cost_per_test": 0.000826,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 76.263,
                "latency": 3.614,
                "stderr": 1.585,
                "cost_per_test": 0.000114,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3.8-max": {
                "accuracy": 76.071,
                "latency": 111.488,
                "stderr": 1.455,
                "cost_per_test": 0.030421,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 75.853,
                "latency": 8.888,
                "stderr": 1.691,
                "cost_per_test": 0.000393,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-4.1-mini-2025-04-14": {
                "accuracy": 75.61,
                "latency": 0.69,
                "stderr": 1.484,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "inception/mercury-2.5": {
                "accuracy": 75.484,
                "latency": 4.523,
                "stderr": 2.0,
                "cost_per_test": 0.000731,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Inception",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514-thinking": {
                "accuracy": 75.086,
                "latency": 11.4,
                "stderr": 1.651,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 74.954,
                "latency": 8.586,
                "stderr": 1.753,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "together/Qwen/Qwen2.5-72B-Instruct-Turbo": {
                "accuracy": 74.874,
                "latency": 1.868,
                "stderr": 2.819,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 74.806,
                "latency": 16.482,
                "stderr": 1.834,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-4o-2024-08-06": {
                "accuracy": 74.718,
                "latency": 1.178,
                "stderr": 2.358,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "minimax/MiniMax-M2.1": {
                "accuracy": 74.551,
                "latency": 21.452,
                "stderr": 1.826,
                "cost_per_test": 0.001587,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "grok/grok-3-mini-fast-low-reasoning": {
                "accuracy": 74.145,
                "latency": 3.991,
                "stderr": 1.671,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "low",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 74.097,
                "latency": 14.158,
                "stderr": 1.56,
                "cost_per_test": 0.006421,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025": {
                "accuracy": 73.981,
                "latency": 0.783,
                "stderr": 1.598,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "fireworks/deepseek-v3": {
                "accuracy": 73.899,
                "latency": 5.227,
                "stderr": 1.648,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "grok/grok-4-0709": {
                "accuracy": 73.869,
                "latency": 53.442,
                "stderr": 1.937,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "together/moonshotai/Kimi-K2-Instruct": {
                "accuracy": 73.649,
                "latency": 2.277,
                "stderr": 1.746,
                "cost_per_test": null,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "cohere/command-a-03-2025": {
                "accuracy": 73.598,
                "latency": 1.385,
                "stderr": 1.556,
                "cost_per_test": null,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 8000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "fireworks/llama4-maverick-instruct-basic": {
                "accuracy": 73.592,
                "latency": 0.806,
                "stderr": 1.499,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "zai/glm-4.5": {
                "accuracy": 73.01,
                "latency": 42.5,
                "stderr": 1.806,
                "cost_per_test": null,
                "temperature": 0.6,
                "top_p": 1.0,
                "max_output_tokens": 81920,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "xiaomi/mimo-v2.5-pro": {
                "accuracy": 72.98,
                "latency": 23.385,
                "stderr": 1.841,
                "cost_per_test": 0.000606,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "minimax/MiniMax-M2.5": {
                "accuracy": 72.55,
                "latency": 27.838,
                "stderr": 1.804,
                "cost_per_test": 0.001806,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "zai/glm-4.6": {
                "accuracy": 71.734,
                "latency": 39.317,
                "stderr": 1.943,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 122880,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "ant/ling-3.0-flash-2607": {
                "accuracy": 71.346,
                "latency": 3.964,
                "stderr": 2.047,
                "cost_per_test": 0.000128,
                "temperature": 0.6,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "grok/grok-4-1-fast-non-reasoning": {
                "accuracy": 70.747,
                "latency": 1.67,
                "stderr": 1.696,
                "cost_per_test": 7.2e-05,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "fireworks/qwen3-235b-a22b": {
                "accuracy": 70.646,
                "latency": 16.92,
                "stderr": 1.674,
                "cost_per_test": null,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "mistralai/mistral-medium-2505": {
                "accuracy": 70.616,
                "latency": 1.81,
                "stderr": 1.751,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "grok/grok-4-fast-non-reasoning": {
                "accuracy": 70.464,
                "latency": 1.19,
                "stderr": 1.822,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "ai21labs/jamba-1.5-large": {
                "accuracy": 70.442,
                "latency": 1.1,
                "stderr": 2.605,
                "cost_per_test": null,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "alibaba/qwen3.8-27b": {
                "accuracy": 70.154,
                "latency": 103.678,
                "stderr": 1.856,
                "cost_per_test": 0.012279,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 69.717,
                "latency": 46.366,
                "stderr": 1.518,
                "cost_per_test": 0.047712,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "together/meta-llama/Llama-4-Scout-17B-16E-Instruct": {
                "accuracy": 69.618,
                "latency": 1.099,
                "stderr": 1.776,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "xiaomi/mimo-v2.5": {
                "accuracy": 69.582,
                "latency": 11.246,
                "stderr": 2.038,
                "cost_per_test": 0.000227,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "cohere/command-r-plus": {
                "accuracy": 69.553,
                "latency": 0.94,
                "stderr": 2.805,
                "cost_per_test": null,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 4000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "together/google/gemma-2-27b-it": {
                "accuracy": 69.21,
                "latency": 2.222,
                "stderr": 2.619,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "ai21labs/jamba-1.5-mini": {
                "accuracy": 69.089,
                "latency": 0.354,
                "stderr": 2.597,
                "cost_per_test": null,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "fireworks/deepseek-v3p2": {
                "accuracy": 68.748,
                "latency": 4.89,
                "stderr": 1.346,
                "cost_per_test": 7e-05,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "anthropic/claude-3-5-haiku-20241022": {
                "accuracy": 68.65,
                "latency": 1.493,
                "stderr": 2.614,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "fireworks/deepseek-v3p2-thinking": {
                "accuracy": 68.646,
                "latency": 31.764,
                "stderr": 1.667,
                "cost_per_test": 0.000434,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "fireworks/gpt-oss-120b": {
                "accuracy": 68.62,
                "latency": 13.18,
                "stderr": 1.891,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "ai21labs/jamba-mini-1.6": {
                "accuracy": 68.312,
                "latency": 0.481,
                "stderr": 1.581,
                "cost_per_test": null,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "ai21labs/jamba-large-1.6": {
                "accuracy": 67.937,
                "latency": 1.542,
                "stderr": 1.157,
                "cost_per_test": null,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "together/meta-llama/Llama-3.3-70B-Instruct-Turbo": {
                "accuracy": 67.065,
                "latency": 0.563,
                "stderr": 3.525,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "openai/gpt-4.1-nano-2025-04-14": {
                "accuracy": 66.738,
                "latency": 0.422,
                "stderr": 2.04,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4-fast-reasoning": {
                "accuracy": 66.411,
                "latency": 3.861,
                "stderr": 1.844,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "mistralai/mistral-small-2503": {
                "accuracy": 66.283,
                "latency": 1.206,
                "stderr": 1.887,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "openai/gpt-3.5-turbo": {
                "accuracy": 66.174,
                "latency": 0.777,
                "stderr": 2.659,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-2.0-flash-001": {
                "accuracy": 64.379,
                "latency": 0.939,
                "stderr": 3.591,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "together/google/gemma-2-9b-it": {
                "accuracy": 63.391,
                "latency": 1.259,
                "stderr": 3.248,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "together/Qwen/Qwen2.5-7B-Instruct-Turbo": {
                "accuracy": 63.301,
                "latency": 0.708,
                "stderr": 3.346,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "poolside/laguna-xs.2": {
                "accuracy": 63.288,
                "latency": 16.793,
                "stderr": 2.145,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "google/gemini-1.5-flash-002": {
                "accuracy": 62.618,
                "latency": 0.772,
                "stderr": 3.456,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "poolside/laguna-m.1": {
                "accuracy": 62.451,
                "latency": 69.822,
                "stderr": 2.269,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "together/meta-llama/Llama-2-70b-hf": {
                "accuracy": 62.305,
                "latency": 6.414,
                "stderr": 3.074,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "together/mistralai/Mixtral-8x7B-v0.1": {
                "accuracy": 62.236,
                "latency": 6.012,
                "stderr": 3.293,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "google/gemini-1.5-pro-002": {
                "accuracy": 61.468,
                "latency": 2.096,
                "stderr": 3.533,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "together/togethercomputer/llama-2-7b": {
                "accuracy": 60.864,
                "latency": 1.204,
                "stderr": 2.929,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "ant/ling-3.0-flash-af-rc3": {
                "accuracy": 60.809,
                "latency": 7.01,
                "stderr": 2.034,
                "cost_per_test": 0.000184,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "openai/o3-mini-2025-01-31": {
                "accuracy": 59.9,
                "latency": 15.631,
                "stderr": 3.205,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "together/togethercomputer/llama-2-13b": {
                "accuracy": 59.559,
                "latency": 1.703,
                "stderr": 3.216,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "fireworks/deepseek-v3-0324": {
                "accuracy": 59.321,
                "latency": 6.029,
                "stderr": 1.487,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "fireworks/gpt-oss-20b": {
                "accuracy": 58.697,
                "latency": 10.493,
                "stderr": 2.266,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "mistralai/magistral-medium-2509": {
                "accuracy": 55.866,
                "latency": 8.795,
                "stderr": 1.921,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "together/mistralai/Mistral-7B-v0.1": {
                "accuracy": 54.826,
                "latency": 0.882,
                "stderr": 3.481,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "google/gemini-1.0-pro-002": {
                "accuracy": 52.382,
                "latency": 0.461,
                "stderr": 3.493,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "cohere/command-a-plus-05-2026": {
                "accuracy": 46.72,
                "latency": 85.57,
                "stderr": 2.408,
                "cost_per_test": 0.123978,
                "temperature": 0.9,
                "top_p": 0.95,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "mistralai/magistral-small-2509": {
                "accuracy": 44.684,
                "latency": 10.701,
                "stderr": 1.913,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "fireworks/deepseek-r1": {
                "accuracy": 42.328,
                "latency": 27.764,
                "stderr": 2.349,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 2048,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "openai/gpt-5-nano-2025-08-07": {
                "accuracy": 42.29,
                "latency": 30.381,
                "stderr": 1.618,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "cohere/command-r": {
                "accuracy": 32.235,
                "latency": 0.485,
                "stderr": 3.169,
                "cost_per_test": null,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 4000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            }
        },
        "conclusion_tasks": {
            "anthropic/claude-fable-5-1": {
                "accuracy": 92.511,
                "latency": 6.903,
                "stderr": 0.79,
                "cost_per_test": 0.023415,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-fable-5": {
                "accuracy": 91.775,
                "latency": 7.255,
                "stderr": 0.667,
                "cost_per_test": 0.019734,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-5": {
                "accuracy": 91.56,
                "latency": 2.373,
                "stderr": 0.696,
                "cost_per_test": 0.005379,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.7-flash": {
                "accuracy": 91.146,
                "latency": 1.884,
                "stderr": 0.891,
                "cost_per_test": 0.002867,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.6-flash": {
                "accuracy": 91.081,
                "latency": 2.694,
                "stderr": 0.662,
                "cost_per_test": 0.003949,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 90.884,
                "latency": 25.395,
                "stderr": 0.601,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-4-argon": {
                "accuracy": 90.847,
                "latency": 9.579,
                "stderr": 0.613,
                "cost_per_test": 0.011481,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.8-flash": {
                "accuracy": 90.691,
                "latency": 2.432,
                "stderr": 0.949,
                "cost_per_test": 0.005226,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.6-sol": {
                "accuracy": 90.492,
                "latency": 2.88,
                "stderr": 0.937,
                "cost_per_test": 0.005618,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4.6": {
                "accuracy": 90.442,
                "latency": 27.344,
                "stderr": 0.885,
                "cost_per_test": 0.01379,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 90.365,
                "latency": 7.663,
                "stderr": 0.595,
                "cost_per_test": 0.009074,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.6-terra": {
                "accuracy": 90.327,
                "latency": 2.373,
                "stderr": 0.916,
                "cost_per_test": 0.001848,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-5.3": {
                "accuracy": 90.305,
                "latency": 12.537,
                "stderr": 0.619,
                "cost_per_test": 0.00386,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 90.168,
                "latency": 15.108,
                "stderr": 0.608,
                "cost_per_test": 0.013817,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-4.7": {
                "accuracy": 90.08,
                "latency": 28.536,
                "stderr": 0.991,
                "cost_per_test": 0.002008,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 90.077,
                "latency": 4.23,
                "stderr": 0.613,
                "cost_per_test": 0.001736,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "kimi/kimi-k3": {
                "accuracy": 90.076,
                "latency": 5.899,
                "stderr": 0.583,
                "cost_per_test": 0.005094,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "kimi/kimi-k2-thinking": {
                "accuracy": 90.006,
                "latency": 52.521,
                "stderr": 1.034,
                "cost_per_test": 0.002334,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "anthropic/claude-opus-4-7": {
                "accuracy": 89.983,
                "latency": 16.717,
                "stderr": 0.934,
                "cost_per_test": 0.007066,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3.5-plus-thinking": {
                "accuracy": 89.939,
                "latency": 30.444,
                "stderr": 0.765,
                "cost_per_test": 0.003255,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "grok/grok-4.3": {
                "accuracy": 89.912,
                "latency": 316.505,
                "stderr": 0.746,
                "cost_per_test": 0.002733,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "alibaba/qwen3.7-max": {
                "accuracy": 89.763,
                "latency": 11.735,
                "stderr": 0.639,
                "cost_per_test": 0.005228,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 89.644,
                "latency": 11.168,
                "stderr": 0.697,
                "cost_per_test": 0.00784,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-5.1": {
                "accuracy": 89.575,
                "latency": 11.322,
                "stderr": 0.723,
                "cost_per_test": 0.001422,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "zai/glm-5.3-flash": {
                "accuracy": 89.573,
                "latency": 6.61,
                "stderr": 0.631,
                "cost_per_test": 0.000144,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "anthropic/claude-sonnet-5": {
                "accuracy": 89.555,
                "latency": 9.309,
                "stderr": 0.753,
                "cost_per_test": 0.009566,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "deepseek/deepseek-v4-pro-0813": {
                "accuracy": 89.548,
                "latency": 17.96,
                "stderr": 0.715,
                "cost_per_test": 0.003565,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "grok/grok-4.5": {
                "accuracy": 89.507,
                "latency": 59.01,
                "stderr": 0.671,
                "cost_per_test": 0.004226,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "meta/muse_spark": {
                "accuracy": 89.47,
                "latency": 28.603,
                "stderr": 0.814,
                "cost_per_test": 0.00096,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 89.436,
                "latency": 7.635,
                "stderr": 0.722,
                "cost_per_test": 0.033545,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 89.425,
                "latency": 7.573,
                "stderr": 0.644,
                "cost_per_test": 0.00546,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 89.413,
                "latency": 52.705,
                "stderr": 0.796,
                "cost_per_test": 0.005986,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 89.339,
                "latency": 8.912,
                "stderr": 0.689,
                "cost_per_test": 0.000458,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "minimax/MiniMax-M3": {
                "accuracy": 89.276,
                "latency": 9.224,
                "stderr": 0.742,
                "cost_per_test": 0.001085,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 512000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 89.257,
                "latency": 10.23,
                "stderr": 0.611,
                "cost_per_test": 0.005926,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 89.177,
                "latency": 20.039,
                "stderr": 0.669,
                "cost_per_test": 0.003064,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 89.117,
                "latency": 115.976,
                "stderr": 0.704,
                "cost_per_test": 0.004307,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "deepseek/deepseek-v4.1-flash": {
                "accuracy": 89.075,
                "latency": 5.698,
                "stderr": 0.782,
                "cost_per_test": 0.001151,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "thinkingmachines/inkling": {
                "accuracy": 88.974,
                "latency": 25.972,
                "stderr": 0.691,
                "cost_per_test": 0.014039,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "anthropic/claude-opus-4-1-20250805": {
                "accuracy": 88.951,
                "latency": 2.637,
                "stderr": 0.817,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 88.939,
                "latency": 18.611,
                "stderr": 0.711,
                "cost_per_test": 0.015054,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-4-8": {
                "accuracy": 88.937,
                "latency": 4.886,
                "stderr": 0.61,
                "cost_per_test": 0.011109,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 88.926,
                "latency": 16.072,
                "stderr": 0.764,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "deepseek/deepseek-v4-flash-0731": {
                "accuracy": 88.892,
                "latency": 3.197,
                "stderr": 0.742,
                "cost_per_test": 0.00036,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "openai/gpt-5.6-luna": {
                "accuracy": 88.864,
                "latency": 3.783,
                "stderr": 0.935,
                "cost_per_test": 0.000524,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-5-thinking": {
                "accuracy": 88.837,
                "latency": 12.629,
                "stderr": 0.723,
                "cost_per_test": 0.001675,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 88.795,
                "latency": 3.904,
                "stderr": 0.703,
                "cost_per_test": 0.003218,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-4.7": {
                "accuracy": 88.636,
                "latency": 29.531,
                "stderr": 0.737,
                "cost_per_test": 0.010857,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "zai/glm-4.5": {
                "accuracy": 88.598,
                "latency": 11.244,
                "stderr": 0.83,
                "cost_per_test": null,
                "temperature": 0.6,
                "top_p": 1.0,
                "max_output_tokens": 81920,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "zai/glm-5.2": {
                "accuracy": 88.587,
                "latency": 4.646,
                "stderr": 0.735,
                "cost_per_test": 0.000935,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "inception/mercury-2.5": {
                "accuracy": 88.542,
                "latency": 3.431,
                "stderr": 0.844,
                "cost_per_test": 0.000651,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Inception",
                "harness": null
            },
            "alibaba/qwen3.5-flash": {
                "accuracy": 88.542,
                "latency": 14.198,
                "stderr": 0.815,
                "cost_per_test": 0.000612,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "tencent/hy4-preview": {
                "accuracy": 88.477,
                "latency": 73.328,
                "stderr": 0.703,
                "cost_per_test": 0.010602,
                "temperature": 0.9,
                "top_p": 1.0,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Tencent",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 88.472,
                "latency": 2.904,
                "stderr": 0.714,
                "cost_per_test": 0.000289,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "thinkingmachines/inkling-small": {
                "accuracy": 88.452,
                "latency": 31.907,
                "stderr": 0.993,
                "cost_per_test": 0.002358,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 262000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "google/gemini-3.5-flash-lite": {
                "accuracy": 88.404,
                "latency": 1.784,
                "stderr": 0.721,
                "cost_per_test": 0.001313,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3.8-27b": {
                "accuracy": 88.4,
                "latency": 51.475,
                "stderr": 0.819,
                "cost_per_test": 0.003936,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "google/gemini-2.5-pro-exp-03-25": {
                "accuracy": 88.312,
                "latency": 4.261,
                "stderr": 0.806,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/o3-2025-04-16": {
                "accuracy": 88.283,
                "latency": 5.751,
                "stderr": 0.797,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3.8-max": {
                "accuracy": 88.23,
                "latency": 26.634,
                "stderr": 0.693,
                "cost_per_test": 0.007137,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514": {
                "accuracy": 88.121,
                "latency": 4.358,
                "stderr": 0.858,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "xiaomi/mimo-v2.5-pro": {
                "accuracy": 88.031,
                "latency": 21.462,
                "stderr": 0.842,
                "cost_per_test": 0.000694,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 87.991,
                "latency": 27.372,
                "stderr": 1.018,
                "cost_per_test": 0.003786,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-4-20250514": {
                "accuracy": 87.942,
                "latency": 3.297,
                "stderr": 0.814,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "fireworks/deepseek-v3p2-thinking": {
                "accuracy": 87.709,
                "latency": 29.776,
                "stderr": 0.815,
                "cost_per_test": 0.000425,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101": {
                "accuracy": 87.7,
                "latency": 2.138,
                "stderr": 0.875,
                "cost_per_test": 0.007212,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-3-mini-fast-high-reasoning": {
                "accuracy": 87.654,
                "latency": 6.571,
                "stderr": 0.895,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-3-7-sonnet-20250219-thinking": {
                "accuracy": 87.587,
                "latency": 6.595,
                "stderr": 0.8,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 5096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "meta/muse_spark_1_2": {
                "accuracy": 87.485,
                "latency": 36.037,
                "stderr": 0.794,
                "cost_per_test": 0.00731,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "nvidia/nemotron-3-ultra-550b-a55b": {
                "accuracy": 87.47,
                "latency": 6.253,
                "stderr": 0.772,
                "cost_per_test": null,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Nvidia",
                "harness": null
            },
            "meta/muse_spark_1_1": {
                "accuracy": 87.328,
                "latency": 18.313,
                "stderr": 0.742,
                "cost_per_test": 0.008371,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "grok/grok-4-0709": {
                "accuracy": 87.31,
                "latency": 30.771,
                "stderr": 0.836,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 87.272,
                "latency": 11.836,
                "stderr": 0.672,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-4o-2024-11-20": {
                "accuracy": 87.208,
                "latency": 0.328,
                "stderr": 0.818,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "fireworks/qwen3-235b-a22b": {
                "accuracy": 87.164,
                "latency": 16.146,
                "stderr": 0.76,
                "cost_per_test": null,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "openai/gpt-4.1-2025-04-14": {
                "accuracy": 87.139,
                "latency": 0.386,
                "stderr": 0.875,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514-thinking": {
                "accuracy": 87.084,
                "latency": 12.413,
                "stderr": 0.881,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "ant/ling-3.0-flash-2607": {
                "accuracy": 87.052,
                "latency": 7.714,
                "stderr": 0.881,
                "cost_per_test": 0.000414,
                "temperature": 0.6,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-04-17": {
                "accuracy": 86.958,
                "latency": 0.447,
                "stderr": 0.881,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/o1-2024-12-17": {
                "accuracy": 86.892,
                "latency": 7.02,
                "stderr": 0.71,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 2048,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4-fast-reasoning": {
                "accuracy": 86.798,
                "latency": 2.854,
                "stderr": 0.797,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 86.698,
                "latency": 8.025,
                "stderr": 0.711,
                "cost_per_test": 0.000348,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025": {
                "accuracy": 86.519,
                "latency": 0.873,
                "stderr": 0.91,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-4o-2024-08-06": {
                "accuracy": 86.442,
                "latency": 0.343,
                "stderr": 0.773,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "xiaomi/mimo-v2.5": {
                "accuracy": 86.386,
                "latency": 8.689,
                "stderr": 0.844,
                "cost_per_test": 0.000189,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025-thinking": {
                "accuracy": 86.297,
                "latency": 3.707,
                "stderr": 0.905,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/o4-mini-2025-04-16": {
                "accuracy": 86.286,
                "latency": 3.623,
                "stderr": 0.799,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 86.259,
                "latency": 13.842,
                "stderr": 1.077,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "minimax/MiniMax-M2.5": {
                "accuracy": 86.241,
                "latency": 13.522,
                "stderr": 0.903,
                "cost_per_test": 0.001249,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "grok/grok-3-mini-fast-low-reasoning": {
                "accuracy": 86.21,
                "latency": 3.246,
                "stderr": 0.921,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "low",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "fireworks/gpt-oss-120b": {
                "accuracy": 86.179,
                "latency": 6.163,
                "stderr": 1.039,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "grok/grok-3": {
                "accuracy": 86.085,
                "latency": 0.379,
                "stderr": 0.944,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025-thinking": {
                "accuracy": 85.782,
                "latency": 1.464,
                "stderr": 0.925,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "together/moonshotai/Kimi-K2-Instruct": {
                "accuracy": 85.674,
                "latency": 0.905,
                "stderr": 0.927,
                "cost_per_test": null,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "zai/glm-4.6": {
                "accuracy": 85.616,
                "latency": 32.329,
                "stderr": 0.861,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 122880,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "poolside/laguna-m.1": {
                "accuracy": 85.459,
                "latency": 45.298,
                "stderr": 0.941,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "alibaba/qwen3-max": {
                "accuracy": 85.392,
                "latency": 0.0,
                "stderr": 0.865,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "minimax/MiniMax-M2.1": {
                "accuracy": 85.318,
                "latency": 9.637,
                "stderr": 0.804,
                "cost_per_test": 0.000695,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 85.107,
                "latency": 7.074,
                "stderr": 0.839,
                "cost_per_test": 0.005368,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "alibaba/qwen3-max-preview": {
                "accuracy": 84.999,
                "latency": 1.451,
                "stderr": 0.861,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/o3-mini-2025-01-31": {
                "accuracy": 84.584,
                "latency": 9.492,
                "stderr": 0.811,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "fireworks/deepseek-v3": {
                "accuracy": 84.255,
                "latency": 5.812,
                "stderr": 0.864,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 84.203,
                "latency": 2.749,
                "stderr": 0.838,
                "cost_per_test": 0.000121,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "fireworks/deepseek-v3-0324": {
                "accuracy": 83.67,
                "latency": 10.726,
                "stderr": 0.927,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "anthropic/claude-3-7-sonnet-20250219": {
                "accuracy": 83.482,
                "latency": 1.108,
                "stderr": 0.826,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "cohere/command-a-03-2025": {
                "accuracy": 83.079,
                "latency": 0.573,
                "stderr": 0.962,
                "cost_per_test": null,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 8000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "together/Qwen/Qwen2.5-72B-Instruct-Turbo": {
                "accuracy": 83.012,
                "latency": 0.513,
                "stderr": 1.026,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "openai/gpt-4-turbo": {
                "accuracy": 82.547,
                "latency": 0.69,
                "stderr": 0.954,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-1.5-pro-002": {
                "accuracy": 82.32,
                "latency": 0.552,
                "stderr": 0.88,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "fireworks/gpt-oss-20b": {
                "accuracy": 82.18,
                "latency": 8.088,
                "stderr": 1.214,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "cohere/command-a-plus-05-2026": {
                "accuracy": 82.12,
                "latency": 9.621,
                "stderr": 1.013,
                "cost_per_test": 0.012736,
                "temperature": 0.9,
                "top_p": 0.95,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "mistralai/magistral-medium-2509": {
                "accuracy": 81.946,
                "latency": 11.895,
                "stderr": 1.013,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "google/gemini-3.5-flash": {
                "accuracy": 81.939,
                "latency": 27.306,
                "stderr": 1.055,
                "cost_per_test": 0.00642,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "fireworks/deepseek-v3p2": {
                "accuracy": 81.856,
                "latency": 1.492,
                "stderr": 0.883,
                "cost_per_test": 3.8e-05,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "grok/grok-4-fast-non-reasoning": {
                "accuracy": 81.733,
                "latency": 0.766,
                "stderr": 0.963,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "fireworks/llama4-maverick-instruct-basic": {
                "accuracy": 81.427,
                "latency": 0.165,
                "stderr": 1.022,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "mistralai/mistral-large-2512": {
                "accuracy": 81.032,
                "latency": 0.793,
                "stderr": 1.001,
                "cost_per_test": 0.000225,
                "temperature": 0.065,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "grok/grok-4-1-fast-non-reasoning": {
                "accuracy": 80.804,
                "latency": 0.589,
                "stderr": 0.945,
                "cost_per_test": 4.7e-05,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-4.1-mini-2025-04-14": {
                "accuracy": 80.674,
                "latency": 0.395,
                "stderr": 1.001,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "mistralai/mistral-medium-2505": {
                "accuracy": 80.586,
                "latency": 0.307,
                "stderr": 0.862,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "google/gemini-2.0-flash-001": {
                "accuracy": 80.388,
                "latency": 0.316,
                "stderr": 0.956,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025": {
                "accuracy": 79.878,
                "latency": 0.359,
                "stderr": 1.051,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "ai21labs/jamba-1.5-large": {
                "accuracy": 79.671,
                "latency": 0.522,
                "stderr": 1.06,
                "cost_per_test": null,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "together/google/gemma-2-27b-it": {
                "accuracy": 79.082,
                "latency": 0.576,
                "stderr": 1.079,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "fireworks/deepseek-r1": {
                "accuracy": 78.983,
                "latency": 21.182,
                "stderr": 0.775,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 2048,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "together/meta-llama/Llama-3.3-70B-Instruct-Turbo": {
                "accuracy": 78.695,
                "latency": 0.606,
                "stderr": 1.01,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "google/gemini-1.5-flash-002": {
                "accuracy": 78.095,
                "latency": 0.36,
                "stderr": 1.028,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "ai21labs/jamba-large-1.6": {
                "accuracy": 77.789,
                "latency": 0.497,
                "stderr": 0.954,
                "cost_per_test": null,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "openai/gpt-5-nano-2025-08-07": {
                "accuracy": 77.278,
                "latency": 25.918,
                "stderr": 1.124,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "poolside/laguna-xs.2": {
                "accuracy": 77.103,
                "latency": 12.554,
                "stderr": 0.915,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "mistralai/magistral-small-2509": {
                "accuracy": 76.791,
                "latency": 5.94,
                "stderr": 0.911,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "anthropic/claude-3-5-haiku-20241022": {
                "accuracy": 76.226,
                "latency": 0.89,
                "stderr": 0.976,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-1.0-pro-002": {
                "accuracy": 74.505,
                "latency": 0.386,
                "stderr": 1.106,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "together/meta-llama/Llama-4-Scout-17B-16E-Instruct": {
                "accuracy": 73.85,
                "latency": 0.226,
                "stderr": 1.086,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "together/meta-llama/Llama-2-70b-hf": {
                "accuracy": 73.2,
                "latency": 11.731,
                "stderr": 1.076,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "ai21labs/jamba-mini-1.6": {
                "accuracy": 73.114,
                "latency": 0.248,
                "stderr": 1.122,
                "cost_per_test": null,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "together/google/gemma-2-9b-it": {
                "accuracy": 71.712,
                "latency": 0.389,
                "stderr": 1.082,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "ai21labs/jamba-1.5-mini": {
                "accuracy": 68.941,
                "latency": 0.238,
                "stderr": 1.138,
                "cost_per_test": null,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "openai/gpt-3.5-turbo": {
                "accuracy": 67.401,
                "latency": 0.345,
                "stderr": 1.13,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-4.1-nano-2025-04-14": {
                "accuracy": 65.648,
                "latency": 0.31,
                "stderr": 1.18,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "together/Qwen/Qwen2.5-7B-Instruct-Turbo": {
                "accuracy": 65.607,
                "latency": 0.339,
                "stderr": 1.3,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "together/mistralai/Mixtral-8x7B-v0.1": {
                "accuracy": 64.321,
                "latency": 10.413,
                "stderr": 1.239,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "mistralai/mistral-small-2503": {
                "accuracy": 61.857,
                "latency": 0.568,
                "stderr": 1.154,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "together/mistralai/Mistral-7B-v0.1": {
                "accuracy": 61.629,
                "latency": 0.527,
                "stderr": 1.08,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "ant/ling-3.0-flash-af-rc3": {
                "accuracy": 61.06,
                "latency": 11.462,
                "stderr": 1.095,
                "cost_per_test": 0.00032,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "together/togethercomputer/llama-2-13b": {
                "accuracy": 57.446,
                "latency": 0.602,
                "stderr": 1.136,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "together/togethercomputer/llama-2-7b": {
                "accuracy": 54.424,
                "latency": 0.408,
                "stderr": 1.003,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "cohere/command-r-plus": {
                "accuracy": 48.306,
                "latency": 0.238,
                "stderr": 1.196,
                "cost_per_test": null,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 4000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "cohere/command-r": {
                "accuracy": 13.98,
                "latency": 0.168,
                "stderr": 0.699,
                "cost_per_test": null,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 4000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            }
        },
        "interpretation_tasks": {
            "google/gemini-3.5-flash": {
                "accuracy": 88.675,
                "latency": 30.374,
                "stderr": 0.289,
                "cost_per_test": 0.006199,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-fable-5-1": {
                "accuracy": 86.902,
                "latency": 10.008,
                "stderr": 0.19,
                "cost_per_test": 0.043747,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-fable-5": {
                "accuracy": 86.863,
                "latency": 8.119,
                "stderr": 0.163,
                "cost_per_test": 0.028526,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-5": {
                "accuracy": 85.84,
                "latency": 3.075,
                "stderr": 0.191,
                "cost_per_test": 0.010305,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.7-flash": {
                "accuracy": 84.608,
                "latency": 2.145,
                "stderr": 0.348,
                "cost_per_test": 0.004299,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.8-flash": {
                "accuracy": 84.589,
                "latency": 4.633,
                "stderr": 0.325,
                "cost_per_test": 0.012231,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.6-sol": {
                "accuracy": 84.398,
                "latency": 7.666,
                "stderr": 0.203,
                "cost_per_test": 0.01698,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 84.251,
                "latency": 9.49,
                "stderr": 0.268,
                "cost_per_test": 0.011919,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-4-argon": {
                "accuracy": 84.097,
                "latency": 12.499,
                "stderr": 0.213,
                "cost_per_test": 0.016184,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "kimi/kimi-k3": {
                "accuracy": 84.092,
                "latency": 5.339,
                "stderr": 0.182,
                "cost_per_test": 0.004089,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "alibaba/qwen3.8-max": {
                "accuracy": 84.061,
                "latency": 22.297,
                "stderr": 0.19,
                "cost_per_test": 0.007109,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 83.921,
                "latency": 22.065,
                "stderr": 0.199,
                "cost_per_test": 0.019498,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 83.908,
                "latency": 5.669,
                "stderr": 0.302,
                "cost_per_test": 0.002582,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.6-flash": {
                "accuracy": 83.894,
                "latency": 2.98,
                "stderr": 0.329,
                "cost_per_test": 0.004921,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 83.836,
                "latency": 15.562,
                "stderr": 0.194,
                "cost_per_test": 0.014325,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 83.823,
                "latency": 9.143,
                "stderr": 0.367,
                "cost_per_test": 0.006094,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-4.6": {
                "accuracy": 83.62,
                "latency": 28.472,
                "stderr": 0.196,
                "cost_per_test": 0.015278,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-opus-4-7": {
                "accuracy": 83.609,
                "latency": 24.393,
                "stderr": 0.204,
                "cost_per_test": 0.007695,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "minimax/MiniMax-M3": {
                "accuracy": 83.511,
                "latency": 6.584,
                "stderr": 0.199,
                "cost_per_test": 0.00083,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 512000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "grok/grok-4.5": {
                "accuracy": 83.22,
                "latency": 65.63,
                "stderr": 0.191,
                "cost_per_test": 0.004689,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-5.6-terra": {
                "accuracy": 83.178,
                "latency": 2.352,
                "stderr": 0.186,
                "cost_per_test": 0.002419,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 83.164,
                "latency": 2.584,
                "stderr": 0.177,
                "cost_per_test": 0.003341,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 83.031,
                "latency": 14.309,
                "stderr": 0.197,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 82.954,
                "latency": 7.524,
                "stderr": 0.186,
                "cost_per_test": 0.03889,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "meta/muse_spark_1_1": {
                "accuracy": 82.876,
                "latency": 14.792,
                "stderr": 0.199,
                "cost_per_test": 0.00777,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 82.756,
                "latency": 12.853,
                "stderr": 0.194,
                "cost_per_test": 0.013416,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-2.5-pro-exp-03-25": {
                "accuracy": 82.736,
                "latency": 3.486,
                "stderr": 0.186,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-4-8": {
                "accuracy": 82.703,
                "latency": 2.245,
                "stderr": 0.202,
                "cost_per_test": 0.007534,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3.5-flash": {
                "accuracy": 82.685,
                "latency": 16.742,
                "stderr": 0.205,
                "cost_per_test": 0.000755,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "grok/grok-4-0709": {
                "accuracy": 82.675,
                "latency": 24.861,
                "stderr": 0.181,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "meta/muse_spark_1_2": {
                "accuracy": 82.542,
                "latency": 27.908,
                "stderr": 0.197,
                "cost_per_test": 0.005359,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 82.469,
                "latency": 19.328,
                "stderr": 0.191,
                "cost_per_test": 0.002893,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 82.416,
                "latency": 6.983,
                "stderr": 0.208,
                "cost_per_test": 0.004895,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "tencent/hy4-preview": {
                "accuracy": 82.334,
                "latency": 76.569,
                "stderr": 0.214,
                "cost_per_test": 0.01148,
                "temperature": 0.9,
                "top_p": 1.0,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Tencent",
                "harness": null
            },
            "alibaba/qwen3.7-max": {
                "accuracy": 82.268,
                "latency": 8.421,
                "stderr": 0.232,
                "cost_per_test": 0.004529,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101": {
                "accuracy": 82.235,
                "latency": 2.149,
                "stderr": 0.188,
                "cost_per_test": 0.012917,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 82.197,
                "latency": 39.793,
                "stderr": 0.198,
                "cost_per_test": 0.005464,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "zai/glm-5.2": {
                "accuracy": 82.17,
                "latency": 3.657,
                "stderr": 0.194,
                "cost_per_test": 0.001164,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "openai/gpt-5.6-luna": {
                "accuracy": 82.142,
                "latency": 4.461,
                "stderr": 0.199,
                "cost_per_test": 0.000696,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 82.126,
                "latency": 12.969,
                "stderr": 0.265,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3.8-27b": {
                "accuracy": 82.031,
                "latency": 32.251,
                "stderr": 0.202,
                "cost_per_test": 0.003654,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 82.006,
                "latency": 17.539,
                "stderr": 0.211,
                "cost_per_test": 0.002808,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "inception/mercury-2.5": {
                "accuracy": 81.911,
                "latency": 3.739,
                "stderr": 0.201,
                "cost_per_test": 0.000802,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Inception",
                "harness": null
            },
            "alibaba/qwen3.5-plus-thinking": {
                "accuracy": 81.869,
                "latency": 32.807,
                "stderr": 0.198,
                "cost_per_test": 0.004016,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "zai/glm-5-thinking": {
                "accuracy": 81.827,
                "latency": 19.112,
                "stderr": 0.229,
                "cost_per_test": 0.002786,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "meta/muse_spark": {
                "accuracy": 81.813,
                "latency": 31.722,
                "stderr": 0.201,
                "cost_per_test": 0.001106,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "grok/grok-4.3": {
                "accuracy": 81.805,
                "latency": 338.953,
                "stderr": 0.203,
                "cost_per_test": 0.003208,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "zai/glm-5.1": {
                "accuracy": 81.709,
                "latency": 14.32,
                "stderr": 0.212,
                "cost_per_test": 0.002309,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 81.653,
                "latency": 8.579,
                "stderr": 0.203,
                "cost_per_test": 0.000625,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-04-17": {
                "accuracy": 81.564,
                "latency": 0.416,
                "stderr": 0.19,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-sonnet-5": {
                "accuracy": 81.412,
                "latency": 2.575,
                "stderr": 0.205,
                "cost_per_test": 0.003312,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "thinkingmachines/inkling": {
                "accuracy": 81.344,
                "latency": 24.324,
                "stderr": 0.2,
                "cost_per_test": 0.01392,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "grok/grok-4.7": {
                "accuracy": 81.264,
                "latency": 25.4,
                "stderr": 0.202,
                "cost_per_test": 0.010767,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/o3-2025-04-16": {
                "accuracy": 81.236,
                "latency": 5.062,
                "stderr": 0.202,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4-fast-reasoning": {
                "accuracy": 80.873,
                "latency": 4.848,
                "stderr": 0.188,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "grok/grok-3": {
                "accuracy": 80.865,
                "latency": 0.404,
                "stderr": 0.195,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "grok/grok-3-mini-fast-high-reasoning": {
                "accuracy": 80.821,
                "latency": 4.869,
                "stderr": 0.201,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-4.1-2025-04-14": {
                "accuracy": 80.768,
                "latency": 0.411,
                "stderr": 0.192,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 80.723,
                "latency": 7.401,
                "stderr": 0.207,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-3-mini-fast-low-reasoning": {
                "accuracy": 80.626,
                "latency": 3.396,
                "stderr": 0.194,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "low",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025-thinking": {
                "accuracy": 80.552,
                "latency": 3.05,
                "stderr": 0.209,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025-thinking": {
                "accuracy": 80.483,
                "latency": 2.31,
                "stderr": 0.194,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "deepseek/deepseek-v4.1-flash": {
                "accuracy": 80.448,
                "latency": 5.273,
                "stderr": 0.203,
                "cost_per_test": 0.001236,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514": {
                "accuracy": 80.416,
                "latency": 4.652,
                "stderr": 0.188,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-4-1-20250805": {
                "accuracy": 80.408,
                "latency": 2.442,
                "stderr": 0.202,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025": {
                "accuracy": 80.385,
                "latency": 1.309,
                "stderr": 0.193,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-4-20250514": {
                "accuracy": 80.346,
                "latency": 3.349,
                "stderr": 0.208,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.5-flash-lite": {
                "accuracy": 80.332,
                "latency": 1.492,
                "stderr": 0.336,
                "cost_per_test": 0.001183,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "kimi/kimi-k2-thinking": {
                "accuracy": 79.98,
                "latency": 42.595,
                "stderr": 0.207,
                "cost_per_test": 0.002114,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514-thinking": {
                "accuracy": 79.902,
                "latency": 10.91,
                "stderr": 0.197,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 79.6,
                "latency": 9.361,
                "stderr": 0.21,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "thinkingmachines/inkling-small": {
                "accuracy": 79.535,
                "latency": 20.598,
                "stderr": 0.197,
                "cost_per_test": 0.001574,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 262000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "zai/glm-5.3-flash": {
                "accuracy": 79.508,
                "latency": 9.846,
                "stderr": 0.211,
                "cost_per_test": 0.000227,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "xiaomi/mimo-v2.5-pro": {
                "accuracy": 79.497,
                "latency": 31.877,
                "stderr": 0.223,
                "cost_per_test": 0.000979,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 79.296,
                "latency": 11.467,
                "stderr": 0.19,
                "cost_per_test": 0.000427,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "fireworks/deepseek-v3-0324": {
                "accuracy": 79.276,
                "latency": 6.96,
                "stderr": 0.201,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "together/moonshotai/Kimi-K2-Instruct": {
                "accuracy": 79.144,
                "latency": 0.903,
                "stderr": 0.202,
                "cost_per_test": null,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "zai/glm-4.7": {
                "accuracy": 79.084,
                "latency": 28.613,
                "stderr": 0.271,
                "cost_per_test": 0.002465,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "zai/glm-5.3": {
                "accuracy": 79.055,
                "latency": 16.245,
                "stderr": 0.215,
                "cost_per_test": 0.005468,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "fireworks/deepseek-v3": {
                "accuracy": 78.976,
                "latency": 4.133,
                "stderr": 0.206,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "google/gemini-2.0-flash-001": {
                "accuracy": 78.938,
                "latency": 0.322,
                "stderr": 0.203,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "zai/glm-4.6": {
                "accuracy": 78.915,
                "latency": 29.207,
                "stderr": 0.212,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 122880,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "together/Qwen/Qwen2.5-72B-Instruct-Turbo": {
                "accuracy": 78.576,
                "latency": 0.592,
                "stderr": 0.2,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "alibaba/qwen3-max": {
                "accuracy": 78.545,
                "latency": 0.0,
                "stderr": 0.2,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 78.293,
                "latency": 2.731,
                "stderr": 0.344,
                "cost_per_test": 0.000309,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "nvidia/nemotron-3-ultra-550b-a55b": {
                "accuracy": 78.266,
                "latency": 6.225,
                "stderr": 0.209,
                "cost_per_test": null,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Nvidia",
                "harness": null
            },
            "openai/o4-mini-2025-04-16": {
                "accuracy": 78.212,
                "latency": 3.557,
                "stderr": 0.208,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3-max-preview": {
                "accuracy": 78.058,
                "latency": 1.54,
                "stderr": 0.205,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "minimax/MiniMax-M2.5": {
                "accuracy": 78.02,
                "latency": 13.91,
                "stderr": 0.204,
                "cost_per_test": 0.001159,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 78.019,
                "latency": 7.293,
                "stderr": 0.268,
                "cost_per_test": 0.005972,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-4o-2024-08-06": {
                "accuracy": 77.968,
                "latency": 0.367,
                "stderr": 0.199,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-4o-2024-11-20": {
                "accuracy": 77.91,
                "latency": 0.325,
                "stderr": 0.209,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "fireworks/qwen3-235b-a22b": {
                "accuracy": 77.681,
                "latency": 11.547,
                "stderr": 0.225,
                "cost_per_test": null,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "grok/grok-4-fast-non-reasoning": {
                "accuracy": 77.668,
                "latency": 1.146,
                "stderr": 0.2,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "grok/grok-4-1-fast-non-reasoning": {
                "accuracy": 77.668,
                "latency": 1.594,
                "stderr": 0.214,
                "cost_per_test": 9e-05,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "deepseek/deepseek-v4-pro-0813": {
                "accuracy": 77.552,
                "latency": 15.491,
                "stderr": 0.237,
                "cost_per_test": 0.003453,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "anthropic/claude-3-7-sonnet-20250219": {
                "accuracy": 77.452,
                "latency": 1.109,
                "stderr": 0.202,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "together/meta-llama/Llama-3.3-70B-Instruct-Turbo": {
                "accuracy": 77.425,
                "latency": 0.56,
                "stderr": 0.194,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "ant/ling-3.0-flash-2607": {
                "accuracy": 77.422,
                "latency": 2.569,
                "stderr": 0.269,
                "cost_per_test": 9.5e-05,
                "temperature": 0.6,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "cohere/command-a-03-2025": {
                "accuracy": 77.414,
                "latency": 0.696,
                "stderr": 0.209,
                "cost_per_test": null,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 8000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "fireworks/gpt-oss-120b": {
                "accuracy": 77.395,
                "latency": 6.506,
                "stderr": 0.245,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "anthropic/claude-3-7-sonnet-20250219-thinking": {
                "accuracy": 77.261,
                "latency": 6.686,
                "stderr": 0.2,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 5096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "fireworks/llama4-maverick-instruct-basic": {
                "accuracy": 76.755,
                "latency": 0.169,
                "stderr": 0.196,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025": {
                "accuracy": 76.624,
                "latency": 0.369,
                "stderr": 0.192,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "mistralai/mistral-large-2512": {
                "accuracy": 76.594,
                "latency": 0.889,
                "stderr": 0.207,
                "cost_per_test": 0.000401,
                "temperature": 0.065,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "openai/gpt-4-turbo": {
                "accuracy": 76.501,
                "latency": 0.939,
                "stderr": 0.233,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-4.1-mini-2025-04-14": {
                "accuracy": 75.92,
                "latency": 0.402,
                "stderr": 0.215,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "fireworks/deepseek-v3p2-thinking": {
                "accuracy": 75.892,
                "latency": 32.315,
                "stderr": 0.231,
                "cost_per_test": 0.000468,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "poolside/laguna-m.1": {
                "accuracy": 75.595,
                "latency": 37.165,
                "stderr": 0.229,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "ai21labs/jamba-large-1.6": {
                "accuracy": 75.377,
                "latency": 0.613,
                "stderr": 0.212,
                "cost_per_test": null,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "openai/o1-2024-12-17": {
                "accuracy": 75.313,
                "latency": 8.514,
                "stderr": 0.213,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 2048,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 75.169,
                "latency": 242.068,
                "stderr": 0.226,
                "cost_per_test": 0.005512,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "anthropic/claude-3-5-haiku-20241022": {
                "accuracy": 74.919,
                "latency": 0.992,
                "stderr": 0.217,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "ai21labs/jamba-1.5-large": {
                "accuracy": 74.878,
                "latency": 0.693,
                "stderr": 0.217,
                "cost_per_test": null,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "xiaomi/mimo-v2.5": {
                "accuracy": 74.351,
                "latency": 12.892,
                "stderr": 0.235,
                "cost_per_test": 0.000317,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "zai/glm-4.5": {
                "accuracy": 74.17,
                "latency": 17.594,
                "stderr": 0.242,
                "cost_per_test": null,
                "temperature": 0.6,
                "top_p": 1.0,
                "max_output_tokens": 81920,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "google/gemini-1.0-pro-002": {
                "accuracy": 73.86,
                "latency": 0.42,
                "stderr": 0.196,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "minimax/MiniMax-M2.1": {
                "accuracy": 73.592,
                "latency": 9.034,
                "stderr": 0.263,
                "cost_per_test": 0.000753,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "together/Qwen/Qwen2.5-7B-Instruct-Turbo": {
                "accuracy": 73.348,
                "latency": 0.313,
                "stderr": 0.248,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "google/gemini-1.5-pro-002": {
                "accuracy": 72.377,
                "latency": 0.57,
                "stderr": 0.217,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "together/meta-llama/Llama-4-Scout-17B-16E-Instruct": {
                "accuracy": 72.037,
                "latency": 0.205,
                "stderr": 0.231,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "cohere/command-r-plus": {
                "accuracy": 71.624,
                "latency": 0.239,
                "stderr": 0.237,
                "cost_per_test": null,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 4000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "mistralai/mistral-small-2503": {
                "accuracy": 71.413,
                "latency": 0.369,
                "stderr": 0.244,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "together/google/gemma-2-9b-it": {
                "accuracy": 71.075,
                "latency": 0.4,
                "stderr": 0.223,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "deepseek/deepseek-v4-flash-0731": {
                "accuracy": 70.993,
                "latency": 3.943,
                "stderr": 0.236,
                "cost_per_test": 0.000592,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "openai/o3-mini-2025-01-31": {
                "accuracy": 70.04,
                "latency": 8.804,
                "stderr": 0.237,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "poolside/laguna-xs.2": {
                "accuracy": 69.947,
                "latency": 12.122,
                "stderr": 0.267,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "fireworks/gpt-oss-20b": {
                "accuracy": 69.903,
                "latency": 6.443,
                "stderr": 0.28,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "ai21labs/jamba-mini-1.6": {
                "accuracy": 69.282,
                "latency": 0.258,
                "stderr": 0.247,
                "cost_per_test": null,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 69.242,
                "latency": 2.539,
                "stderr": 0.266,
                "cost_per_test": 0.00011,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "ai21labs/jamba-1.5-mini": {
                "accuracy": 69.181,
                "latency": 0.249,
                "stderr": 0.252,
                "cost_per_test": null,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "fireworks/deepseek-r1": {
                "accuracy": 66.411,
                "latency": 19.442,
                "stderr": 0.234,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 2048,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "openai/gpt-3.5-turbo": {
                "accuracy": 65.515,
                "latency": 0.46,
                "stderr": 0.21,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "together/google/gemma-2-27b-it": {
                "accuracy": 65.128,
                "latency": 0.544,
                "stderr": 0.229,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "fireworks/deepseek-v3p2": {
                "accuracy": 62.948,
                "latency": 1.526,
                "stderr": 0.215,
                "cost_per_test": 8.7e-05,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "mistralai/mistral-medium-2505": {
                "accuracy": 62.799,
                "latency": 0.385,
                "stderr": 0.257,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "ant/ling-3.0-flash-af-rc3": {
                "accuracy": 62.613,
                "latency": 5.478,
                "stderr": 0.29,
                "cost_per_test": 0.00018,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "google/gemini-1.5-flash-002": {
                "accuracy": 61.847,
                "latency": 0.404,
                "stderr": 0.208,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "together/meta-llama/Llama-2-70b-hf": {
                "accuracy": 58.277,
                "latency": 11.031,
                "stderr": 0.136,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "openai/gpt-4.1-nano-2025-04-14": {
                "accuracy": 57.822,
                "latency": 0.334,
                "stderr": 0.254,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "cohere/command-a-plus-05-2026": {
                "accuracy": 54.446,
                "latency": 7.838,
                "stderr": 0.288,
                "cost_per_test": 0.011865,
                "temperature": 0.9,
                "top_p": 0.95,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "together/mistralai/Mixtral-8x7B-v0.1": {
                "accuracy": 52.164,
                "latency": 12.31,
                "stderr": 0.12,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "mistralai/magistral-medium-2509": {
                "accuracy": 52.125,
                "latency": 16.249,
                "stderr": 0.309,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "together/togethercomputer/llama-2-13b": {
                "accuracy": 51.063,
                "latency": 1.101,
                "stderr": 0.169,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "together/mistralai/Mistral-7B-v0.1": {
                "accuracy": 49.631,
                "latency": 1.919,
                "stderr": 0.109,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "together/togethercomputer/llama-2-7b": {
                "accuracy": 48.545,
                "latency": 0.742,
                "stderr": 0.128,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "openai/gpt-5-nano-2025-08-07": {
                "accuracy": 47.563,
                "latency": 23.351,
                "stderr": 0.284,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "cohere/command-r": {
                "accuracy": 38.716,
                "latency": 0.186,
                "stderr": 0.208,
                "cost_per_test": null,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 4000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "mistralai/magistral-small-2509": {
                "accuracy": 31.417,
                "latency": 6.782,
                "stderr": 0.196,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            }
        },
        "rhetoric_tasks": {
            "google/gemini-3.1-pro-preview": {
                "accuracy": 84.136,
                "latency": 9.066,
                "stderr": 0.766,
                "cost_per_test": 0.006197,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 84.09,
                "latency": 7.44,
                "stderr": 0.815,
                "cost_per_test": 0.009151,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 83.87,
                "latency": 4.036,
                "stderr": 0.882,
                "cost_per_test": 0.001805,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-fable-5": {
                "accuracy": 83.772,
                "latency": 6.836,
                "stderr": 0.839,
                "cost_per_test": 0.020967,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.5-flash": {
                "accuracy": 83.384,
                "latency": 29.001,
                "stderr": 1.281,
                "cost_per_test": 0.006559,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 82.946,
                "latency": 4.484,
                "stderr": 0.917,
                "cost_per_test": 0.00244,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4.6": {
                "accuracy": 82.884,
                "latency": 21.236,
                "stderr": 0.789,
                "cost_per_test": 0.011671,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 82.872,
                "latency": 4.813,
                "stderr": 0.916,
                "cost_per_test": 0.005559,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-4-argon": {
                "accuracy": 82.838,
                "latency": 13.383,
                "stderr": 1.191,
                "cost_per_test": 0.017171,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 82.824,
                "latency": 14.027,
                "stderr": 1.304,
                "cost_per_test": 0.012138,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4.5": {
                "accuracy": 82.802,
                "latency": 61.924,
                "stderr": 0.808,
                "cost_per_test": 0.003947,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "zai/glm-5.3": {
                "accuracy": 82.752,
                "latency": 5.187,
                "stderr": 0.817,
                "cost_per_test": 0.001903,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "openai/gpt-5.6-luna": {
                "accuracy": 82.747,
                "latency": 2.7,
                "stderr": 0.782,
                "cost_per_test": 0.000284,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4-0709": {
                "accuracy": 82.683,
                "latency": 12.169,
                "stderr": 0.847,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-3.7-flash": {
                "accuracy": 82.53,
                "latency": 2.02,
                "stderr": 1.146,
                "cost_per_test": 0.003787,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "minimax/MiniMax-M3": {
                "accuracy": 82.455,
                "latency": 6.409,
                "stderr": 0.866,
                "cost_per_test": 0.000697,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 512000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 82.448,
                "latency": 7.139,
                "stderr": 1.076,
                "cost_per_test": 0.006809,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-fable-5-1": {
                "accuracy": 82.295,
                "latency": 5.555,
                "stderr": 1.227,
                "cost_per_test": 0.023252,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-04-17": {
                "accuracy": 82.207,
                "latency": 0.423,
                "stderr": 0.824,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "meta/muse_spark_1_2": {
                "accuracy": 82.182,
                "latency": 22.426,
                "stderr": 1.19,
                "cost_per_test": 0.00488,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "meta/muse_spark_1_1": {
                "accuracy": 82.147,
                "latency": 12.902,
                "stderr": 1.152,
                "cost_per_test": 0.006965,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "openai/gpt-5.6-terra": {
                "accuracy": 82.098,
                "latency": 2.004,
                "stderr": 1.094,
                "cost_per_test": 0.002097,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-2.5-pro-exp-03-25": {
                "accuracy": 82.094,
                "latency": 2.732,
                "stderr": 0.871,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.6-sol": {
                "accuracy": 82.054,
                "latency": 2.608,
                "stderr": 1.12,
                "cost_per_test": 0.00629,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "kimi/kimi-k3": {
                "accuracy": 81.933,
                "latency": 4.958,
                "stderr": 1.159,
                "cost_per_test": 0.003016,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "google/gemini-3.6-flash": {
                "accuracy": 81.702,
                "latency": 2.452,
                "stderr": 1.103,
                "cost_per_test": 0.004255,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3.7-max": {
                "accuracy": 81.65,
                "latency": 9.665,
                "stderr": 0.87,
                "cost_per_test": 0.005424,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "tencent/hy4-preview": {
                "accuracy": 81.641,
                "latency": 59.183,
                "stderr": 1.053,
                "cost_per_test": 0.008926,
                "temperature": 0.9,
                "top_p": 1.0,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Tencent",
                "harness": null
            },
            "grok/grok-3-mini-fast-high-reasoning": {
                "accuracy": 81.61,
                "latency": 4.084,
                "stderr": 0.894,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "grok/grok-3": {
                "accuracy": 81.552,
                "latency": 0.408,
                "stderr": 0.865,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/o3-2025-04-16": {
                "accuracy": 81.512,
                "latency": 3.629,
                "stderr": 0.866,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4.7": {
                "accuracy": 81.501,
                "latency": 11.779,
                "stderr": 1.076,
                "cost_per_test": 0.005892,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 81.482,
                "latency": 2.953,
                "stderr": 0.9,
                "cost_per_test": 0.003578,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3-max": {
                "accuracy": 81.432,
                "latency": 0.0,
                "stderr": 0.769,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025-thinking": {
                "accuracy": 81.406,
                "latency": 1.43,
                "stderr": 0.719,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 81.4,
                "latency": 23.913,
                "stderr": 1.179,
                "cost_per_test": 0.003209,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "alibaba/qwen3.8-27b": {
                "accuracy": 81.379,
                "latency": 33.935,
                "stderr": 1.07,
                "cost_per_test": 0.001619,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "grok/grok-3-mini-fast-low-reasoning": {
                "accuracy": 81.366,
                "latency": 3.278,
                "stderr": 0.861,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "low",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-3.8-flash": {
                "accuracy": 81.333,
                "latency": 3.436,
                "stderr": 1.35,
                "cost_per_test": 0.009015,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-4-fast-reasoning": {
                "accuracy": 81.265,
                "latency": 2.779,
                "stderr": 0.837,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "zai/glm-5.2": {
                "accuracy": 81.263,
                "latency": 3.041,
                "stderr": 0.99,
                "cost_per_test": 0.001005,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "alibaba/qwen3-max-preview": {
                "accuracy": 81.229,
                "latency": 1.569,
                "stderr": 0.83,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "alibaba/qwen3.5-plus-thinking": {
                "accuracy": 81.221,
                "latency": 29.649,
                "stderr": 0.887,
                "cost_per_test": 0.003689,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 81.152,
                "latency": 7.275,
                "stderr": 1.212,
                "cost_per_test": 0.00033,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "together/moonshotai/Kimi-K2-Instruct": {
                "accuracy": 81.128,
                "latency": 0.894,
                "stderr": 0.801,
                "cost_per_test": null,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "zai/glm-5.1": {
                "accuracy": 81.108,
                "latency": 12.873,
                "stderr": 0.889,
                "cost_per_test": 0.001795,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025-thinking": {
                "accuracy": 81.104,
                "latency": 2.768,
                "stderr": 0.84,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "meta/muse_spark": {
                "accuracy": 81.101,
                "latency": 28.855,
                "stderr": 0.886,
                "cost_per_test": 0.001036,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "anthropic/claude-sonnet-5": {
                "accuracy": 81.07,
                "latency": 2.558,
                "stderr": 1.105,
                "cost_per_test": 0.003374,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "fireworks/deepseek-v3": {
                "accuracy": 81.062,
                "latency": 4.108,
                "stderr": 0.853,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "zai/glm-5.3-flash": {
                "accuracy": 81.046,
                "latency": 3.911,
                "stderr": 1.018,
                "cost_per_test": 0.000104,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 81.033,
                "latency": 16.284,
                "stderr": 0.878,
                "cost_per_test": 0.002717,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025": {
                "accuracy": 81.031,
                "latency": 0.841,
                "stderr": 0.795,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 80.968,
                "latency": 9.146,
                "stderr": 0.854,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-4o-2024-11-20": {
                "accuracy": 80.954,
                "latency": 0.343,
                "stderr": 0.817,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-4o-2024-08-06": {
                "accuracy": 80.878,
                "latency": 0.353,
                "stderr": 0.777,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 80.804,
                "latency": 6.991,
                "stderr": 0.933,
                "cost_per_test": 0.000552,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "anthropic/claude-opus-5": {
                "accuracy": 80.788,
                "latency": 2.543,
                "stderr": 1.169,
                "cost_per_test": 0.008611,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "fireworks/llama4-maverick-instruct-basic": {
                "accuracy": 80.766,
                "latency": 0.162,
                "stderr": 0.819,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 80.725,
                "latency": 9.747,
                "stderr": 1.149,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "zai/glm-5-thinking": {
                "accuracy": 80.649,
                "latency": 14.362,
                "stderr": 0.918,
                "cost_per_test": 0.002135,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "grok/grok-4.3": {
                "accuracy": 80.606,
                "latency": 347.169,
                "stderr": 0.877,
                "cost_per_test": 0.002656,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "grok/grok-4-1-fast-non-reasoning": {
                "accuracy": 80.595,
                "latency": 1.036,
                "stderr": 1.025,
                "cost_per_test": 8.5e-05,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-3.5-flash-lite": {
                "accuracy": 80.595,
                "latency": 1.379,
                "stderr": 1.109,
                "cost_per_test": 0.001105,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3.5-flash": {
                "accuracy": 80.564,
                "latency": 16.071,
                "stderr": 0.883,
                "cost_per_test": 0.00076,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "thinkingmachines/inkling": {
                "accuracy": 80.563,
                "latency": 11.39,
                "stderr": 1.321,
                "cost_per_test": 0.006985,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 80.514,
                "latency": 2.351,
                "stderr": 0.833,
                "cost_per_test": 0.000262,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-4-fast-non-reasoning": {
                "accuracy": 80.446,
                "latency": 0.559,
                "stderr": 0.892,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "inception/mercury-2.5": {
                "accuracy": 80.314,
                "latency": 4.26,
                "stderr": 0.861,
                "cost_per_test": 0.00084,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Inception",
                "harness": null
            },
            "openai/gpt-4.1-2025-04-14": {
                "accuracy": 80.218,
                "latency": 0.4,
                "stderr": 0.834,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-2.0-flash-001": {
                "accuracy": 80.211,
                "latency": 0.324,
                "stderr": 1.04,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-4-20250514": {
                "accuracy": 80.211,
                "latency": 3.602,
                "stderr": 1.162,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 80.184,
                "latency": 6.208,
                "stderr": 0.914,
                "cost_per_test": 0.032262,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "fireworks/deepseek-v3-0324": {
                "accuracy": 80.163,
                "latency": 6.982,
                "stderr": 0.942,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "deepseek/deepseek-v4.1-flash": {
                "accuracy": 80.127,
                "latency": 3.183,
                "stderr": 1.169,
                "cost_per_test": 0.000611,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "openai/gpt-4-turbo": {
                "accuracy": 80.12,
                "latency": 0.95,
                "stderr": 0.997,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-4-1-20250805": {
                "accuracy": 80.038,
                "latency": 2.416,
                "stderr": 1.168,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514": {
                "accuracy": 79.985,
                "latency": 4.709,
                "stderr": 0.836,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "thinkingmachines/inkling-small": {
                "accuracy": 79.795,
                "latency": 13.32,
                "stderr": 1.104,
                "cost_per_test": 0.000966,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 262000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101": {
                "accuracy": 79.777,
                "latency": 2.187,
                "stderr": 1.202,
                "cost_per_test": 0.013139,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "deepseek/deepseek-v4-pro-0813": {
                "accuracy": 79.739,
                "latency": 6.75,
                "stderr": 1.161,
                "cost_per_test": 0.001615,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "zai/glm-4.7": {
                "accuracy": 79.7,
                "latency": 23.862,
                "stderr": 1.228,
                "cost_per_test": 0.002035,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "cohere/command-a-03-2025": {
                "accuracy": 79.598,
                "latency": 0.57,
                "stderr": 0.826,
                "cost_per_test": null,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 8000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514-thinking": {
                "accuracy": 79.558,
                "latency": 8.476,
                "stderr": 1.148,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "together/meta-llama/Llama-3.3-70B-Instruct-Turbo": {
                "accuracy": 79.51,
                "latency": 0.467,
                "stderr": 1.023,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 79.469,
                "latency": 27.051,
                "stderr": 1.346,
                "cost_per_test": 0.002685,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-3-7-sonnet-20250219": {
                "accuracy": 79.422,
                "latency": 1.084,
                "stderr": 1.055,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "fireworks/qwen3-235b-a22b": {
                "accuracy": 79.389,
                "latency": 8.378,
                "stderr": 1.003,
                "cost_per_test": null,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "anthropic/claude-3-7-sonnet-20250219-thinking": {
                "accuracy": 79.366,
                "latency": 5.542,
                "stderr": 1.252,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 5096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3.8-max": {
                "accuracy": 79.047,
                "latency": 12.546,
                "stderr": 1.197,
                "cost_per_test": 0.003856,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 78.957,
                "latency": 4.951,
                "stderr": 0.955,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "minimax/MiniMax-M2.5": {
                "accuracy": 78.835,
                "latency": 8.925,
                "stderr": 1.025,
                "cost_per_test": 0.000914,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "openai/gpt-4.1-mini-2025-04-14": {
                "accuracy": 78.775,
                "latency": 0.437,
                "stderr": 0.836,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-4-7": {
                "accuracy": 78.722,
                "latency": 17.819,
                "stderr": 1.206,
                "cost_per_test": 0.006968,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 78.634,
                "latency": 7.66,
                "stderr": 1.119,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-4-8": {
                "accuracy": 78.389,
                "latency": 2.188,
                "stderr": 1.376,
                "cost_per_test": 0.006984,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "together/Qwen/Qwen2.5-72B-Instruct-Turbo": {
                "accuracy": 78.36,
                "latency": 0.545,
                "stderr": 1.173,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025": {
                "accuracy": 78.108,
                "latency": 0.366,
                "stderr": 0.907,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "mistralai/mistral-large-2512": {
                "accuracy": 77.926,
                "latency": 0.886,
                "stderr": 1.238,
                "cost_per_test": 0.000427,
                "temperature": 0.065,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 77.909,
                "latency": 7.072,
                "stderr": 1.114,
                "cost_per_test": 0.005725,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/o1-2024-12-17": {
                "accuracy": 77.891,
                "latency": 8.145,
                "stderr": 1.248,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 2048,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "nvidia/nemotron-3-ultra-550b-a55b": {
                "accuracy": 77.885,
                "latency": 3.366,
                "stderr": 1.088,
                "cost_per_test": null,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Nvidia",
                "harness": null
            },
            "together/meta-llama/Llama-4-Scout-17B-16E-Instruct": {
                "accuracy": 77.837,
                "latency": 0.194,
                "stderr": 0.91,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "google/gemini-1.0-pro-002": {
                "accuracy": 77.78,
                "latency": 0.436,
                "stderr": 0.848,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "xiaomi/mimo-v2.5": {
                "accuracy": 77.776,
                "latency": 10.166,
                "stderr": 1.043,
                "cost_per_test": 0.000254,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "zai/glm-4.5": {
                "accuracy": 77.641,
                "latency": 6.622,
                "stderr": 1.311,
                "cost_per_test": null,
                "temperature": 0.6,
                "top_p": 1.0,
                "max_output_tokens": 81920,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "openai/o3-mini-2025-01-31": {
                "accuracy": 77.53,
                "latency": 11.706,
                "stderr": 1.143,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-4.6": {
                "accuracy": 77.244,
                "latency": 31.96,
                "stderr": 1.334,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 122880,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 77.085,
                "latency": 2.308,
                "stderr": 0.925,
                "cost_per_test": 8.8e-05,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "ant/ling-3.0-flash-2607": {
                "accuracy": 76.702,
                "latency": 2.429,
                "stderr": 1.093,
                "cost_per_test": 0.000104,
                "temperature": 0.6,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "cohere/command-r-plus": {
                "accuracy": 75.635,
                "latency": 0.247,
                "stderr": 0.842,
                "cost_per_test": null,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 4000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 75.316,
                "latency": 209.987,
                "stderr": 1.259,
                "cost_per_test": 0.003008,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "ant/ling-3.0-flash-af-rc3": {
                "accuracy": 75.257,
                "latency": 4.309,
                "stderr": 1.032,
                "cost_per_test": 0.000143,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "ai21labs/jamba-large-1.6": {
                "accuracy": 74.37,
                "latency": 0.567,
                "stderr": 1.27,
                "cost_per_test": null,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "poolside/laguna-m.1": {
                "accuracy": 73.756,
                "latency": 30.517,
                "stderr": 1.419,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "anthropic/claude-3-5-haiku-20241022": {
                "accuracy": 73.208,
                "latency": 0.954,
                "stderr": 1.439,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "kimi/kimi-k2-thinking": {
                "accuracy": 73.203,
                "latency": 33.295,
                "stderr": 0.998,
                "cost_per_test": 0.001787,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "ai21labs/jamba-1.5-large": {
                "accuracy": 73.026,
                "latency": 0.714,
                "stderr": 1.292,
                "cost_per_test": null,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "together/Qwen/Qwen2.5-7B-Instruct-Turbo": {
                "accuracy": 72.957,
                "latency": 0.293,
                "stderr": 1.303,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "together/google/gemma-2-9b-it": {
                "accuracy": 72.476,
                "latency": 0.388,
                "stderr": 1.071,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "fireworks/deepseek-v3p2-thinking": {
                "accuracy": 72.096,
                "latency": 29.918,
                "stderr": 1.389,
                "cost_per_test": 0.000433,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "minimax/MiniMax-M2.1": {
                "accuracy": 71.806,
                "latency": 7.218,
                "stderr": 1.597,
                "cost_per_test": 0.000605,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "together/google/gemma-2-27b-it": {
                "accuracy": 71.531,
                "latency": 0.587,
                "stderr": 1.451,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "cohere/command-r": {
                "accuracy": 71.005,
                "latency": 0.163,
                "stderr": 1.375,
                "cost_per_test": null,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 4000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "fireworks/gpt-oss-120b": {
                "accuracy": 70.893,
                "latency": 8.044,
                "stderr": 1.158,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "openai/o4-mini-2025-04-16": {
                "accuracy": 70.599,
                "latency": 3.466,
                "stderr": 1.493,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "deepseek/deepseek-v4-flash-0731": {
                "accuracy": 70.442,
                "latency": 2.898,
                "stderr": 1.479,
                "cost_per_test": 0.000351,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "mistralai/mistral-small-2503": {
                "accuracy": 69.778,
                "latency": 0.443,
                "stderr": 1.004,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "fireworks/gpt-oss-20b": {
                "accuracy": 69.296,
                "latency": 4.381,
                "stderr": 1.379,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "fireworks/deepseek-r1": {
                "accuracy": 68.729,
                "latency": 11.562,
                "stderr": 0.728,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 2048,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "cohere/command-a-plus-05-2026": {
                "accuracy": 68.648,
                "latency": 4.851,
                "stderr": 1.472,
                "cost_per_test": 0.007517,
                "temperature": 0.9,
                "top_p": 0.95,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "poolside/laguna-xs.2": {
                "accuracy": 67.98,
                "latency": 8.765,
                "stderr": 1.354,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "ai21labs/jamba-mini-1.6": {
                "accuracy": 65.935,
                "latency": 0.259,
                "stderr": 1.471,
                "cost_per_test": null,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "openai/gpt-3.5-turbo": {
                "accuracy": 65.714,
                "latency": 0.465,
                "stderr": 1.095,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-1.5-pro-002": {
                "accuracy": 65.108,
                "latency": 0.599,
                "stderr": 0.666,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "xiaomi/mimo-v2.5-pro": {
                "accuracy": 64.964,
                "latency": 26.638,
                "stderr": 1.677,
                "cost_per_test": 0.000894,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "ai21labs/jamba-1.5-mini": {
                "accuracy": 64.682,
                "latency": 0.253,
                "stderr": 1.548,
                "cost_per_test": null,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "together/meta-llama/Llama-2-70b-hf": {
                "accuracy": 64.494,
                "latency": 12.007,
                "stderr": 1.369,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "openai/gpt-4.1-nano-2025-04-14": {
                "accuracy": 62.804,
                "latency": 0.353,
                "stderr": 1.596,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-1.5-flash-002": {
                "accuracy": 61.657,
                "latency": 0.397,
                "stderr": 0.667,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "fireworks/deepseek-v3p2": {
                "accuracy": 61.532,
                "latency": 1.51,
                "stderr": 1.26,
                "cost_per_test": 8.2e-05,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "mistralai/mistral-medium-2505": {
                "accuracy": 59.773,
                "latency": 0.407,
                "stderr": 1.086,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "together/mistralai/Mistral-7B-v0.1": {
                "accuracy": 51.087,
                "latency": 1.699,
                "stderr": 1.364,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "together/togethercomputer/llama-2-13b": {
                "accuracy": 50.156,
                "latency": 1.64,
                "stderr": 1.314,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "together/mistralai/Mixtral-8x7B-v0.1": {
                "accuracy": 46.708,
                "latency": 13.914,
                "stderr": 1.126,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "together/togethercomputer/llama-2-7b": {
                "accuracy": 45.218,
                "latency": 0.774,
                "stderr": 0.487,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "mistralai/magistral-medium-2509": {
                "accuracy": 43.556,
                "latency": 8.779,
                "stderr": 1.499,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "openai/gpt-5-nano-2025-08-07": {
                "accuracy": 30.641,
                "latency": 14.893,
                "stderr": 0.839,
                "cost_per_test": null,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "mistralai/magistral-small-2509": {
                "accuracy": 27.28,
                "latency": 3.618,
                "stderr": 0.755,
                "cost_per_test": null,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            }
        }
    }
}