{
    "metadata": {
        "benchmark": "Finance Agent v1.1",
        "slug": "finance_agent",
        "description": "Evaluating agents on core financial analyst tasks",
        "benchmark_id": "finance_agent",
        "family": "finance_agent",
        "version": "1.1",
        "updated": "2026-06-04",
        "dataset_type": "private",
        "industry": "finance",
        "tasks": {
            "overall": "Overall",
            "simple_retrieval_quantitative": "Simple retrieval - Quantitative",
            "simple_retrieval_qualitative": "Simple retrieval - Qualitative",
            "complex_retrieval": "Complex Retrieval",
            "numerical_reasoning": "Numerical Reasoning",
            "financial_modeling": "Financial Modeling  Projections",
            "market_analysis": "Market Analysis",
            "beat_or_miss": "Beat or Miss",
            "trends": "Trends",
            "adjustments": "Adjustments",
            "vals_index_subset": "Vals Index Subset"
        },
        "models": [
            "ai21labs/jamba-large-1.7",
            "alibaba/qwen3-max",
            "alibaba/qwen3.5-flash",
            "alibaba/qwen3.5-plus-thinking",
            "alibaba/qwen3.6-max-preview",
            "alibaba/qwen3.6-plus",
            "anthropic/claude-haiku-4-5-20251001-thinking",
            "anthropic/claude-opus-4-5-20251101-thinking",
            "anthropic/claude-opus-4-6-thinking",
            "anthropic/claude-opus-4-7",
            "anthropic/claude-sonnet-4-5-20250929-thinking",
            "anthropic/claude-sonnet-4-6",
            "cohere/command-a-03-2025",
            "deepseek/deepseek-v4-pro",
            "fireworks/deepseek-v3p2",
            "fireworks/deepseek-v3p2-thinking",
            "fireworks/gpt-oss-120b",
            "google/gemini-2.5-pro",
            "google/gemini-3-flash-preview",
            "google/gemini-3-pro-preview",
            "google/gemini-3.1-flash-lite-preview",
            "google/gemini-3.1-pro-preview",
            "google/gemma-4-31b-it",
            "grok/grok-4-0709",
            "grok/grok-4-1-fast-non-reasoning",
            "grok/grok-4-1-fast-reasoning",
            "grok/grok-4-fast-reasoning",
            "grok/grok-4.20-0309-reasoning",
            "grok/grok-4.3",
            "kimi/kimi-k2-thinking",
            "kimi/kimi-k2.5-thinking",
            "kimi/kimi-k2.6",
            "meta/muse_spark",
            "minimax/MiniMax-M2.1",
            "minimax/MiniMax-M2.5",
            "minimax/MiniMax-M2.7",
            "mistralai/mistral-large-2512",
            "mistralai/mistral-medium-3.5",
            "openai/gpt-4o-2024-08-06",
            "openai/gpt-5-2025-08-07",
            "openai/gpt-5-mini-2025-08-07",
            "openai/gpt-5.1-2025-11-13",
            "openai/gpt-5.2-2025-12-11",
            "openai/gpt-5.4-2026-03-05",
            "openai/gpt-5.4-mini-2026-03-17",
            "openai/gpt-5.4-nano-2026-03-17",
            "openai/gpt-5.5",
            "zai/glm-4.6",
            "zai/glm-4.7",
            "zai/glm-5-thinking",
            "zai/glm-5.1"
        ],
        "partners": [],
        "showBadge": false,
        "visible": true,
        "use_cost_per_test": true,
        "runner": "custom",
        "mode": "agentic",
        "archived": true,
        "total_models": 51
    },
    "tasks": {
        "overall": {
            "anthropic/claude-opus-4-7": {
                "accuracy": 64.373,
                "latency": 271.055,
                "stderr": 2.793,
                "cost_per_test": 0.802032,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 63.331,
                "latency": 348.995,
                "stderr": 2.841,
                "cost_per_test": 1.436698,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "meta/muse_spark": {
                "accuracy": 60.595,
                "latency": 418.139,
                "stderr": 2.839,
                "cost_per_test": 0.063451,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 60.389,
                "latency": 588.439,
                "stderr": 2.796,
                "cost_per_test": 0.585364,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 60.046,
                "latency": 289.732,
                "stderr": 2.775,
                "cost_per_test": 1.108198,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 59.963,
                "latency": 855.159,
                "stderr": 2.835,
                "cost_per_test": 1.32831,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 59.717,
                "latency": 265.722,
                "stderr": 2.804,
                "cost_per_test": 0.872622,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 58.81,
                "latency": 181.87,
                "stderr": 2.812,
                "cost_per_test": 1.501497,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 58.535,
                "latency": 585.414,
                "stderr": 2.87,
                "cost_per_test": 0.977848,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-5.1": {
                "accuracy": 57.655,
                "latency": 501.808,
                "stderr": 2.797,
                "cost_per_test": 0.291851,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 57.152,
                "latency": 656.674,
                "stderr": 2.853,
                "cost_per_test": 1.412332,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 57.056,
                "latency": 1505.876,
                "stderr": 2.856,
                "cost_per_test": 0.492709,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 55.309,
                "latency": 578.064,
                "stderr": 2.802,
                "cost_per_test": 0.473742,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 55.154,
                "latency": 183.624,
                "stderr": 2.798,
                "cost_per_test": 0.558567,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 54.627,
                "latency": 327.762,
                "stderr": 2.889,
                "cost_per_test": 0.151866,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 54.5,
                "latency": 202.068,
                "stderr": 2.86,
                "cost_per_test": 1.102422,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3.5-plus-thinking": {
                "accuracy": 54.475,
                "latency": 360.478,
                "stderr": 2.894,
                "cost_per_test": 0.236035,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "grok/grok-4.3": {
                "accuracy": 53.812,
                "latency": 795.917,
                "stderr": 2.828,
                "cost_per_test": 0.442142,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "grok/grok-4-0709": {
                "accuracy": 53.506,
                "latency": 321.035,
                "stderr": 2.854,
                "cost_per_test": 1.074806,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 53.405,
                "latency": 876.254,
                "stderr": 2.837,
                "cost_per_test": 0.491653,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-5-thinking": {
                "accuracy": 53.182,
                "latency": 564.093,
                "stderr": 2.798,
                "cost_per_test": 0.501332,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "alibaba/qwen3.6-max-preview": {
                "accuracy": 52.785,
                "latency": 1838.57,
                "stderr": 2.87,
                "cost_per_test": 0.588068,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 52.448,
                "latency": 90.582,
                "stderr": 2.805,
                "cost_per_test": 0.058346,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 52.295,
                "latency": 127.028,
                "stderr": 0.0,
                "cost_per_test": 0.362389,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 52.151,
                "latency": 926.611,
                "stderr": 2.879,
                "cost_per_test": 0.586458,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 51.928,
                "latency": 643.508,
                "stderr": 2.897,
                "cost_per_test": 0.142449,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemma-4-31b-it": {
                "accuracy": 50.788,
                "latency": 5360.333,
                "stderr": 2.856,
                "cost_per_test": 0.0,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 50.622,
                "latency": 268.752,
                "stderr": 2.845,
                "cost_per_test": 0.178983,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 48.402,
                "latency": 1292.755,
                "stderr": 2.791,
                "cost_per_test": 0.160333,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 47.801,
                "latency": 283.802,
                "stderr": 2.828,
                "cost_per_test": 0.012919,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 47.598,
                "latency": 876.089,
                "stderr": 2.784,
                "cost_per_test": 0.369606,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 46.931,
                "latency": 117.937,
                "stderr": 2.849,
                "cost_per_test": 0.377285,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 46.123,
                "latency": 102.442,
                "stderr": 2.81,
                "cost_per_test": 0.071759,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "mistralai/mistral-medium-3.5": {
                "accuracy": 46.113,
                "latency": 562.191,
                "stderr": 2.883,
                "cost_per_test": 1.147222,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "grok/grok-4-fast-reasoning": {
                "accuracy": 46.084,
                "latency": 58.799,
                "stderr": 2.841,
                "cost_per_test": 0.046595,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "zai/glm-4.7": {
                "accuracy": 45.977,
                "latency": 605.003,
                "stderr": 2.818,
                "cost_per_test": 0.244376,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "alibaba/qwen3.5-flash": {
                "accuracy": 45.639,
                "latency": 248.888,
                "stderr": 2.775,
                "cost_per_test": 0.054887,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "grok/grok-4-1-fast-non-reasoning": {
                "accuracy": 44.362,
                "latency": 93.211,
                "stderr": 2.738,
                "cost_per_test": 0.089182,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "alibaba/qwen3-max": {
                "accuracy": 44.295,
                "latency": 315.362,
                "stderr": 2.806,
                "cost_per_test": 0.801603,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "google/gemini-2.5-pro": {
                "accuracy": 41.589,
                "latency": 1839.558,
                "stderr": 2.783,
                "cost_per_test": 0.34118,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "minimax/MiniMax-M2.5": {
                "accuracy": 38.579,
                "latency": 1355.194,
                "stderr": 2.646,
                "cost_per_test": 0.164656,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "kimi/kimi-k2-thinking": {
                "accuracy": 36.647,
                "latency": 666.531,
                "stderr": 2.609,
                "cost_per_test": 0.249634,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "zai/glm-4.6": {
                "accuracy": 36.48,
                "latency": 736.029,
                "stderr": 2.792,
                "cost_per_test": 0.187068,
                "temperature": 0.6,
                "top_p": 1.0,
                "max_output_tokens": 122880,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "minimax/MiniMax-M2.1": {
                "accuracy": 33.35,
                "latency": 1301.289,
                "stderr": 2.58,
                "cost_per_test": 0.114177,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "fireworks/gpt-oss-120b": {
                "accuracy": 21.541,
                "latency": 185.074,
                "stderr": 2.238,
                "cost_per_test": 0.064574,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "mistralai/mistral-large-2512": {
                "accuracy": 18.049,
                "latency": 91.152,
                "stderr": 2.176,
                "cost_per_test": 0.086052,
                "temperature": 0.065,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "openai/gpt-4o-2024-08-06": {
                "accuracy": 8.064,
                "latency": 31.141,
                "stderr": 1.496,
                "cost_per_test": 0.158124,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "cohere/command-a-03-2025": {
                "accuracy": 4.226,
                "latency": 573.215,
                "stderr": 0.966,
                "cost_per_test": 2.339409,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 8000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "fireworks/deepseek-v3p2-thinking": {
                "accuracy": 2.345,
                "latency": 1721.463,
                "stderr": 0.771,
                "cost_per_test": 0.208831,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 20480,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "ai21labs/jamba-large-1.7": {
                "accuracy": 0.37,
                "latency": 2330.85,
                "stderr": 0.364,
                "cost_per_test": 5.040786,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "fireworks/deepseek-v3p2": {
                "accuracy": 0.0,
                "latency": 100.617,
                "stderr": 0.0,
                "cost_per_test": 0.03968,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 20480,
                "reasoning": null,
                "reasoning_effort": "none",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            }
        },
        "simple_retrieval_quantitative": {
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 84.375,
                "latency": 129.669,
                "stderr": 4.575,
                "cost_per_test": 1.2877,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 82.812,
                "latency": 153.519,
                "stderr": 4.753,
                "cost_per_test": 1.093441,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 82.812,
                "latency": 216.258,
                "stderr": 4.753,
                "cost_per_test": 0.153689,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "anthropic/claude-opus-4-7": {
                "accuracy": 82.812,
                "latency": 250.828,
                "stderr": 4.753,
                "cost_per_test": 1.065722,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 81.25,
                "latency": 179.664,
                "stderr": 4.917,
                "cost_per_test": 0.954565,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "meta/muse_spark": {
                "accuracy": 81.25,
                "latency": 252.719,
                "stderr": 4.917,
                "cost_per_test": 0.04243,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "alibaba/qwen3.5-plus-thinking": {
                "accuracy": 81.25,
                "latency": 252.837,
                "stderr": 4.917,
                "cost_per_test": 0.166958,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 81.25,
                "latency": 502.874,
                "stderr": 4.917,
                "cost_per_test": 0.650193,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 81.25,
                "latency": 1259.381,
                "stderr": 4.917,
                "cost_per_test": 0.411207,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "alibaba/qwen3.6-max-preview": {
                "accuracy": 81.25,
                "latency": 1714.608,
                "stderr": 4.917,
                "cost_per_test": 0.475902,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 79.688,
                "latency": 207.101,
                "stderr": 5.069,
                "cost_per_test": 0.101333,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 79.688,
                "latency": 215.786,
                "stderr": 5.069,
                "cost_per_test": 0.709311,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "zai/glm-5-thinking": {
                "accuracy": 79.688,
                "latency": 397.958,
                "stderr": 5.069,
                "cost_per_test": 0.370901,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 79.688,
                "latency": 478.401,
                "stderr": 5.069,
                "cost_per_test": 0.983137,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4-1-fast-non-reasoning": {
                "accuracy": 78.125,
                "latency": 77.708,
                "stderr": 5.208,
                "cost_per_test": 0.081731,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 78.125,
                "latency": 146.447,
                "stderr": 5.208,
                "cost_per_test": 0.805399,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "zai/glm-5.1": {
                "accuracy": 78.125,
                "latency": 236.381,
                "stderr": 5.208,
                "cost_per_test": 0.195437,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 78.125,
                "latency": 456.973,
                "stderr": 5.208,
                "cost_per_test": 0.791379,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 78.125,
                "latency": 688.309,
                "stderr": 5.208,
                "cost_per_test": 2.696609,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 76.562,
                "latency": 67.287,
                "stderr": 5.337,
                "cost_per_test": 0.04102,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 76.562,
                "latency": 136.297,
                "stderr": 5.337,
                "cost_per_test": 0.544308,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3.5-flash": {
                "accuracy": 76.562,
                "latency": 228.947,
                "stderr": 5.337,
                "cost_per_test": 0.044851,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "grok/grok-4-0709": {
                "accuracy": 76.562,
                "latency": 330.774,
                "stderr": 5.337,
                "cost_per_test": 0.957088,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 76.562,
                "latency": 734.568,
                "stderr": 5.337,
                "cost_per_test": 0.38756,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemma-4-31b-it": {
                "accuracy": 76.562,
                "latency": 4830.239,
                "stderr": 5.337,
                "cost_per_test": 0.0,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 75.0,
                "latency": 83.372,
                "stderr": 5.455,
                "cost_per_test": 0.33239,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 75.0,
                "latency": 195.187,
                "stderr": 5.455,
                "cost_per_test": 0.340643,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 75.0,
                "latency": 294.455,
                "stderr": 5.455,
                "cost_per_test": 0.049297,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-4.7": {
                "accuracy": 75.0,
                "latency": 443.538,
                "stderr": 5.455,
                "cost_per_test": 0.208039,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "grok/grok-4.3": {
                "accuracy": 75.0,
                "latency": 727.55,
                "stderr": 5.455,
                "cost_per_test": 0.380697,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 75.0,
                "latency": 982.016,
                "stderr": 5.455,
                "cost_per_test": 0.122708,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "alibaba/qwen3-max": {
                "accuracy": 73.438,
                "latency": 244.65,
                "stderr": 5.564,
                "cost_per_test": 0.6787,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 73.438,
                "latency": 302.026,
                "stderr": 5.564,
                "cost_per_test": 0.380354,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 73.438,
                "latency": 488.088,
                "stderr": 5.564,
                "cost_per_test": 0.109191,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4-fast-reasoning": {
                "accuracy": 71.875,
                "latency": 49.277,
                "stderr": 5.665,
                "cost_per_test": 0.042327,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 70.312,
                "latency": 1060.479,
                "stderr": 5.756,
                "cost_per_test": 0.437138,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 68.75,
                "latency": 44.464,
                "stderr": 5.84,
                "cost_per_test": 0.061291,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "zai/glm-4.6": {
                "accuracy": 68.75,
                "latency": 519.068,
                "stderr": 5.84,
                "cost_per_test": 0.120604,
                "temperature": 0.6,
                "top_p": 1.0,
                "max_output_tokens": 122880,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "kimi/kimi-k2-thinking": {
                "accuracy": 68.75,
                "latency": 639.899,
                "stderr": 5.84,
                "cost_per_test": 0.237098,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 67.188,
                "latency": 71.002,
                "stderr": 5.916,
                "cost_per_test": 0.289161,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "mistralai/mistral-medium-3.5": {
                "accuracy": 67.188,
                "latency": 469.832,
                "stderr": 5.916,
                "cost_per_test": 0.92905,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "google/gemini-2.5-pro": {
                "accuracy": 65.625,
                "latency": 69.151,
                "stderr": 5.984,
                "cost_per_test": 0.271475,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "minimax/MiniMax-M2.5": {
                "accuracy": 64.062,
                "latency": 1561.78,
                "stderr": 6.045,
                "cost_per_test": 0.146045,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "minimax/MiniMax-M2.1": {
                "accuracy": 57.812,
                "latency": 940.084,
                "stderr": 6.222,
                "cost_per_test": 0.084633,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "fireworks/gpt-oss-120b": {
                "accuracy": 48.438,
                "latency": 128.256,
                "stderr": 6.296,
                "cost_per_test": 0.04326,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "mistralai/mistral-large-2512": {
                "accuracy": 26.562,
                "latency": 69.915,
                "stderr": 5.564,
                "cost_per_test": 0.084149,
                "temperature": 0.065,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "openai/gpt-4o-2024-08-06": {
                "accuracy": 17.188,
                "latency": 24.535,
                "stderr": 4.753,
                "cost_per_test": 0.172287,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "cohere/command-a-03-2025": {
                "accuracy": 9.375,
                "latency": 278.952,
                "stderr": 3.672,
                "cost_per_test": 1.216857,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 8000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "fireworks/deepseek-v3p2-thinking": {
                "accuracy": 7.812,
                "latency": 1627.847,
                "stderr": 3.381,
                "cost_per_test": 0.271341,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 20480,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "fireworks/deepseek-v3p2": {
                "accuracy": 0.0,
                "latency": 81.619,
                "stderr": 0.0,
                "cost_per_test": 0.026806,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 20480,
                "reasoning": null,
                "reasoning_effort": "none",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "ai21labs/jamba-large-1.7": {
                "accuracy": 0.0,
                "latency": 2129.522,
                "stderr": 0.0,
                "cost_per_test": 4.636133,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            }
        },
        "simple_retrieval_qualitative": {
            "anthropic/claude-opus-4-7": {
                "accuracy": 77.049,
                "latency": 254.906,
                "stderr": 5.429,
                "cost_per_test": 0.973579,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "meta/muse_spark": {
                "accuracy": 73.77,
                "latency": 424.489,
                "stderr": 5.679,
                "cost_per_test": 0.047441,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 72.131,
                "latency": 139.162,
                "stderr": 5.788,
                "cost_per_test": 0.950788,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 72.131,
                "latency": 264.485,
                "stderr": 5.788,
                "cost_per_test": 1.042422,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 72.131,
                "latency": 798.657,
                "stderr": 5.788,
                "cost_per_test": 2.959409,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 70.492,
                "latency": 176.405,
                "stderr": 5.888,
                "cost_per_test": 0.502268,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 70.492,
                "latency": 402.817,
                "stderr": 5.888,
                "cost_per_test": 0.657803,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 70.492,
                "latency": 407.135,
                "stderr": 5.888,
                "cost_per_test": 0.498508,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 68.852,
                "latency": 175.299,
                "stderr": 5.979,
                "cost_per_test": 0.738057,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 67.213,
                "latency": 157.453,
                "stderr": 6.06,
                "cost_per_test": 0.679758,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 67.213,
                "latency": 487.628,
                "stderr": 6.06,
                "cost_per_test": 0.3342,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4.3": {
                "accuracy": 67.213,
                "latency": 749.039,
                "stderr": 6.06,
                "cost_per_test": 0.322221,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 67.213,
                "latency": 1419.626,
                "stderr": 6.06,
                "cost_per_test": 0.649727,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-5.1": {
                "accuracy": 65.574,
                "latency": 447.37,
                "stderr": 6.134,
                "cost_per_test": 0.183406,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 65.574,
                "latency": 954.872,
                "stderr": 6.134,
                "cost_per_test": 0.240064,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "alibaba/qwen3.5-plus-thinking": {
                "accuracy": 62.295,
                "latency": 285.883,
                "stderr": 6.257,
                "cost_per_test": 0.171434,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "zai/glm-5-thinking": {
                "accuracy": 62.295,
                "latency": 476.817,
                "stderr": 6.257,
                "cost_per_test": 0.392072,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 62.295,
                "latency": 614.695,
                "stderr": 6.257,
                "cost_per_test": 1.103721,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 60.656,
                "latency": 386.203,
                "stderr": 6.307,
                "cost_per_test": 0.114641,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 60.656,
                "latency": 566.711,
                "stderr": 6.307,
                "cost_per_test": 0.112713,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 59.016,
                "latency": 75.078,
                "stderr": 6.349,
                "cost_per_test": 0.364638,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 59.016,
                "latency": 509.015,
                "stderr": 6.349,
                "cost_per_test": 0.217292,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3.6-max-preview": {
                "accuracy": 59.016,
                "latency": 2103.795,
                "stderr": 6.349,
                "cost_per_test": 0.388644,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 57.377,
                "latency": 100.679,
                "stderr": 6.384,
                "cost_per_test": 0.234283,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 57.377,
                "latency": 1458.67,
                "stderr": 6.384,
                "cost_per_test": 0.107036,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 55.738,
                "latency": 71.773,
                "stderr": 6.412,
                "cost_per_test": 0.043066,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "grok/grok-4-0709": {
                "accuracy": 55.738,
                "latency": 158.559,
                "stderr": 6.412,
                "cost_per_test": 0.585651,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 55.738,
                "latency": 160.711,
                "stderr": 6.412,
                "cost_per_test": 0.100733,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 55.738,
                "latency": 636.699,
                "stderr": 6.412,
                "cost_per_test": 0.35451,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemma-4-31b-it": {
                "accuracy": 55.738,
                "latency": 4538.12,
                "stderr": 6.412,
                "cost_per_test": 0.0,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-4-fast-reasoning": {
                "accuracy": 54.098,
                "latency": 40.853,
                "stderr": 6.433,
                "cost_per_test": 0.028495,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 54.098,
                "latency": 81.817,
                "stderr": 6.433,
                "cost_per_test": 0.241203,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 54.098,
                "latency": 265.461,
                "stderr": 6.433,
                "cost_per_test": 0.049297,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4-1-fast-non-reasoning": {
                "accuracy": 52.459,
                "latency": 72.462,
                "stderr": 6.447,
                "cost_per_test": 0.063411,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "alibaba/qwen3.5-flash": {
                "accuracy": 52.459,
                "latency": 161.764,
                "stderr": 6.447,
                "cost_per_test": 0.03707,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "alibaba/qwen3-max": {
                "accuracy": 52.459,
                "latency": 222.163,
                "stderr": 6.447,
                "cost_per_test": 0.438528,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "minimax/MiniMax-M2.5": {
                "accuracy": 50.82,
                "latency": 662.339,
                "stderr": 6.454,
                "cost_per_test": 0.091211,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "mistralai/mistral-medium-3.5": {
                "accuracy": 49.18,
                "latency": 412.575,
                "stderr": 6.454,
                "cost_per_test": 0.790002,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "zai/glm-4.7": {
                "accuracy": 45.902,
                "latency": 486.086,
                "stderr": 6.433,
                "cost_per_test": 0.17263,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 42.623,
                "latency": 130.982,
                "stderr": 6.384,
                "cost_per_test": 0.043699,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-2.5-pro": {
                "accuracy": 40.984,
                "latency": 1353.971,
                "stderr": 6.349,
                "cost_per_test": 0.206071,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "minimax/MiniMax-M2.1": {
                "accuracy": 39.344,
                "latency": 577.161,
                "stderr": 6.307,
                "cost_per_test": 0.071424,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "kimi/kimi-k2-thinking": {
                "accuracy": 37.705,
                "latency": 589.757,
                "stderr": 6.257,
                "cost_per_test": 0.219743,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "zai/glm-4.6": {
                "accuracy": 34.426,
                "latency": 506.02,
                "stderr": 6.134,
                "cost_per_test": 0.120052,
                "temperature": 0.6,
                "top_p": 1.0,
                "max_output_tokens": 122880,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "fireworks/gpt-oss-120b": {
                "accuracy": 26.23,
                "latency": 148.404,
                "stderr": 5.679,
                "cost_per_test": 0.041772,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "mistralai/mistral-large-2512": {
                "accuracy": 24.59,
                "latency": 67.274,
                "stderr": 5.559,
                "cost_per_test": 0.049884,
                "temperature": 0.065,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "openai/gpt-4o-2024-08-06": {
                "accuracy": 14.754,
                "latency": 21.935,
                "stderr": 4.578,
                "cost_per_test": 0.098134,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "cohere/command-a-03-2025": {
                "accuracy": 11.475,
                "latency": 506.479,
                "stderr": 4.115,
                "cost_per_test": 1.929011,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 8000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "fireworks/deepseek-v3p2-thinking": {
                "accuracy": 1.639,
                "latency": 1079.71,
                "stderr": 1.639,
                "cost_per_test": 0.167332,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 20480,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "fireworks/deepseek-v3p2": {
                "accuracy": 0.0,
                "latency": 85.644,
                "stderr": 0.0,
                "cost_per_test": 0.029892,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 20480,
                "reasoning": null,
                "reasoning_effort": "none",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "ai21labs/jamba-large-1.7": {
                "accuracy": 0.0,
                "latency": 1906.031,
                "stderr": 0.0,
                "cost_per_test": 4.247687,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            }
        },
        "complex_retrieval": {
            "meta/muse_spark": {
                "accuracy": 61.111,
                "latency": 794.332,
                "stderr": 11.824,
                "cost_per_test": 0.076937,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 61.111,
                "latency": 798.277,
                "stderr": 11.824,
                "cost_per_test": 1.528404,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-4-7": {
                "accuracy": 55.556,
                "latency": 186.326,
                "stderr": 12.052,
                "cost_per_test": 1.886095,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 55.556,
                "latency": 342.844,
                "stderr": 12.052,
                "cost_per_test": 1.56232,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 55.556,
                "latency": 518.591,
                "stderr": 12.052,
                "cost_per_test": 0.968083,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 55.556,
                "latency": 707.884,
                "stderr": 12.052,
                "cost_per_test": 0.186076,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 50.0,
                "latency": 74.362,
                "stderr": 12.127,
                "cost_per_test": 0.516798,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 50.0,
                "latency": 299.21,
                "stderr": 12.127,
                "cost_per_test": 1.23424,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 50.0,
                "latency": 320.345,
                "stderr": 12.127,
                "cost_per_test": 1.198148,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 50.0,
                "latency": 356.34,
                "stderr": 12.127,
                "cost_per_test": 0.223263,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "alibaba/qwen3.5-plus-thinking": {
                "accuracy": 50.0,
                "latency": 417.396,
                "stderr": 12.127,
                "cost_per_test": 0.290746,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "zai/glm-5.1": {
                "accuracy": 50.0,
                "latency": 774.993,
                "stderr": 12.127,
                "cost_per_test": 0.399531,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 50.0,
                "latency": 1578.85,
                "stderr": 12.127,
                "cost_per_test": 0.623184,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "grok/grok-4-fast-reasoning": {
                "accuracy": 44.444,
                "latency": 72.387,
                "stderr": 12.052,
                "cost_per_test": 0.067252,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 44.444,
                "latency": 131.923,
                "stderr": 12.052,
                "cost_per_test": 0.862279,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 44.444,
                "latency": 369.454,
                "stderr": 12.052,
                "cost_per_test": 0.770135,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 44.444,
                "latency": 747.497,
                "stderr": 12.052,
                "cost_per_test": 1.636396,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 44.444,
                "latency": 1058.915,
                "stderr": 12.052,
                "cost_per_test": 6.678619,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 44.444,
                "latency": 1737.217,
                "stderr": 12.052,
                "cost_per_test": 0.276719,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 38.889,
                "latency": 47.108,
                "stderr": 11.824,
                "cost_per_test": 0.108504,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 38.889,
                "latency": 115.406,
                "stderr": 11.824,
                "cost_per_test": 0.419958,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 38.889,
                "latency": 241.013,
                "stderr": 11.824,
                "cost_per_test": 1.508435,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 38.889,
                "latency": 264.521,
                "stderr": 11.824,
                "cost_per_test": 0.049297,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 38.889,
                "latency": 276.685,
                "stderr": 11.824,
                "cost_per_test": 0.212545,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "mistralai/mistral-medium-3.5": {
                "accuracy": 38.889,
                "latency": 688.639,
                "stderr": 11.824,
                "cost_per_test": 1.420662,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "grok/grok-4.3": {
                "accuracy": 38.889,
                "latency": 700.774,
                "stderr": 11.824,
                "cost_per_test": 0.534092,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 38.889,
                "latency": 1096.312,
                "stderr": 11.824,
                "cost_per_test": 0.828715,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 38.889,
                "latency": 1097.072,
                "stderr": 11.824,
                "cost_per_test": 0.790685,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-5-thinking": {
                "accuracy": 38.889,
                "latency": 1137.898,
                "stderr": 11.824,
                "cost_per_test": 1.13938,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "alibaba/qwen3.6-max-preview": {
                "accuracy": 38.889,
                "latency": 1817.215,
                "stderr": 11.824,
                "cost_per_test": 0.684503,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "zai/glm-4.6": {
                "accuracy": 38.889,
                "latency": 2040.85,
                "stderr": 11.824,
                "cost_per_test": 0.393365,
                "temperature": 0.6,
                "top_p": 1.0,
                "max_output_tokens": 122880,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "grok/grok-4-1-fast-non-reasoning": {
                "accuracy": 33.333,
                "latency": 84.909,
                "stderr": 11.433,
                "cost_per_test": 0.104927,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 33.333,
                "latency": 208.278,
                "stderr": 11.433,
                "cost_per_test": 1.969786,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3.5-flash": {
                "accuracy": 33.333,
                "latency": 308.574,
                "stderr": 11.433,
                "cost_per_test": 0.068515,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "grok/grok-4-0709": {
                "accuracy": 33.333,
                "latency": 399.865,
                "stderr": 11.433,
                "cost_per_test": 1.871865,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "zai/glm-4.7": {
                "accuracy": 33.333,
                "latency": 1091.095,
                "stderr": 11.433,
                "cost_per_test": 0.425376,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "minimax/MiniMax-M2.5": {
                "accuracy": 33.333,
                "latency": 1282.536,
                "stderr": 11.433,
                "cost_per_test": 0.267115,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 27.778,
                "latency": 103.108,
                "stderr": 10.863,
                "cost_per_test": 0.095085,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "alibaba/qwen3-max": {
                "accuracy": 27.778,
                "latency": 383.244,
                "stderr": 10.863,
                "cost_per_test": 1.26127,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "google/gemma-4-31b-it": {
                "accuracy": 27.778,
                "latency": 6789.1,
                "stderr": 10.863,
                "cost_per_test": 0.0,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-2.5-pro": {
                "accuracy": 22.222,
                "latency": 90.497,
                "stderr": 10.083,
                "cost_per_test": 0.444738,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "fireworks/gpt-oss-120b": {
                "accuracy": 22.222,
                "latency": 252.839,
                "stderr": 10.083,
                "cost_per_test": 0.086433,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 22.222,
                "latency": 805.849,
                "stderr": 10.083,
                "cost_per_test": 0.537775,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "minimax/MiniMax-M2.1": {
                "accuracy": 22.222,
                "latency": 1601.765,
                "stderr": 10.083,
                "cost_per_test": 0.140671,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "kimi/kimi-k2-thinking": {
                "accuracy": 16.667,
                "latency": 741.955,
                "stderr": 9.039,
                "cost_per_test": 0.354127,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "openai/gpt-4o-2024-08-06": {
                "accuracy": 11.111,
                "latency": 30.532,
                "stderr": 7.622,
                "cost_per_test": 0.172964,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "mistralai/mistral-large-2512": {
                "accuracy": 11.111,
                "latency": 78.717,
                "stderr": 7.622,
                "cost_per_test": 0.063092,
                "temperature": 0.065,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "fireworks/deepseek-v3p2": {
                "accuracy": 0.0,
                "latency": 183.866,
                "stderr": 0.0,
                "cost_per_test": 0.088637,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 20480,
                "reasoning": null,
                "reasoning_effort": "none",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "cohere/command-a-03-2025": {
                "accuracy": 0.0,
                "latency": 621.527,
                "stderr": 0.0,
                "cost_per_test": 3.26366,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 8000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "fireworks/deepseek-v3p2-thinking": {
                "accuracy": 0.0,
                "latency": 1131.1,
                "stderr": 0.0,
                "cost_per_test": 0.192951,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 20480,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "ai21labs/jamba-large-1.7": {
                "accuracy": 0.0,
                "latency": 2547.236,
                "stderr": 0.0,
                "cost_per_test": 5.59086,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            }
        },
        "numerical_reasoning": {
            "anthropic/claude-opus-4-7": {
                "accuracy": 75.0,
                "latency": 137.166,
                "stderr": 6.063,
                "cost_per_test": 1.2368,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 75.0,
                "latency": 279.53,
                "stderr": 6.063,
                "cost_per_test": 0.953255,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 73.077,
                "latency": 362.438,
                "stderr": 6.211,
                "cost_per_test": 1.444306,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 73.077,
                "latency": 505.346,
                "stderr": 6.211,
                "cost_per_test": 1.083191,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 73.077,
                "latency": 558.723,
                "stderr": 6.211,
                "cost_per_test": 0.753221,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 73.077,
                "latency": 1578.519,
                "stderr": 6.211,
                "cost_per_test": 0.539183,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 71.154,
                "latency": 261.55,
                "stderr": 6.344,
                "cost_per_test": 1.087519,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-4-0709": {
                "accuracy": 71.154,
                "latency": 275.418,
                "stderr": 6.344,
                "cost_per_test": 1.092803,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "meta/muse_spark": {
                "accuracy": 71.154,
                "latency": 333.324,
                "stderr": 6.344,
                "cost_per_test": 0.059055,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 71.154,
                "latency": 585.369,
                "stderr": 6.344,
                "cost_per_test": 0.789753,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 71.154,
                "latency": 711.612,
                "stderr": 6.344,
                "cost_per_test": 2.644793,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 69.231,
                "latency": 382.719,
                "stderr": 6.463,
                "cost_per_test": 0.481592,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4.3": {
                "accuracy": 69.231,
                "latency": 861.506,
                "stderr": 6.463,
                "cost_per_test": 0.516914,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 67.308,
                "latency": 79.763,
                "stderr": 6.569,
                "cost_per_test": 0.370981,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 67.308,
                "latency": 236.51,
                "stderr": 6.569,
                "cost_per_test": 0.620674,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "zai/glm-5.1": {
                "accuracy": 67.308,
                "latency": 401.302,
                "stderr": 6.569,
                "cost_per_test": 0.282513,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 67.308,
                "latency": 856.084,
                "stderr": 6.569,
                "cost_per_test": 0.369333,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 65.385,
                "latency": 88.779,
                "stderr": 6.662,
                "cost_per_test": 0.0652,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 65.385,
                "latency": 171.193,
                "stderr": 6.662,
                "cost_per_test": 1.467632,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 65.385,
                "latency": 256.76,
                "stderr": 6.662,
                "cost_per_test": 0.049297,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3.5-plus-thinking": {
                "accuracy": 65.385,
                "latency": 400.245,
                "stderr": 6.662,
                "cost_per_test": 0.225417,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "zai/glm-5-thinking": {
                "accuracy": 65.385,
                "latency": 438.001,
                "stderr": 6.662,
                "cost_per_test": 0.415083,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 65.385,
                "latency": 600.302,
                "stderr": 6.662,
                "cost_per_test": 0.383004,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 65.385,
                "latency": 606.848,
                "stderr": 6.662,
                "cost_per_test": 0.417218,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 63.462,
                "latency": 108.485,
                "stderr": 6.743,
                "cost_per_test": 0.078406,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 63.462,
                "latency": 186.189,
                "stderr": 6.743,
                "cost_per_test": 0.934477,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 63.462,
                "latency": 286.07,
                "stderr": 6.743,
                "cost_per_test": 0.112421,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "alibaba/qwen3.5-flash": {
                "accuracy": 63.462,
                "latency": 302.097,
                "stderr": 6.743,
                "cost_per_test": 0.066926,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "zai/glm-4.7": {
                "accuracy": 63.462,
                "latency": 555.15,
                "stderr": 6.743,
                "cost_per_test": 0.22697,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 63.462,
                "latency": 616.352,
                "stderr": 6.743,
                "cost_per_test": 0.12539,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3.6-max-preview": {
                "accuracy": 63.462,
                "latency": 1814.392,
                "stderr": 6.743,
                "cost_per_test": 0.509877,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "google/gemma-4-31b-it": {
                "accuracy": 63.462,
                "latency": 5438.408,
                "stderr": 6.743,
                "cost_per_test": 0.0,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 61.538,
                "latency": 245.625,
                "stderr": 6.812,
                "cost_per_test": 0.189084,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 61.538,
                "latency": 1484.45,
                "stderr": 6.812,
                "cost_per_test": 0.15207,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "grok/grok-4-fast-reasoning": {
                "accuracy": 59.615,
                "latency": 55.388,
                "stderr": 6.871,
                "cost_per_test": 0.048871,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-2.5-pro": {
                "accuracy": 59.615,
                "latency": 4626.506,
                "stderr": 6.871,
                "cost_per_test": 0.344628,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-4-1-fast-non-reasoning": {
                "accuracy": 57.692,
                "latency": 91.138,
                "stderr": 6.918,
                "cost_per_test": 0.09138,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 57.692,
                "latency": 125.348,
                "stderr": 6.918,
                "cost_per_test": 0.370984,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "mistralai/mistral-medium-3.5": {
                "accuracy": 57.692,
                "latency": 550.349,
                "stderr": 6.918,
                "cost_per_test": 1.157369,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "kimi/kimi-k2-thinking": {
                "accuracy": 55.769,
                "latency": 522.55,
                "stderr": 6.955,
                "cost_per_test": 0.205628,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "alibaba/qwen3-max": {
                "accuracy": 53.846,
                "latency": 304.473,
                "stderr": 6.981,
                "cost_per_test": 0.868295,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "minimax/MiniMax-M2.5": {
                "accuracy": 48.077,
                "latency": 512.485,
                "stderr": 6.996,
                "cost_per_test": 0.128606,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "minimax/MiniMax-M2.1": {
                "accuracy": 40.385,
                "latency": 1615.528,
                "stderr": 6.871,
                "cost_per_test": 0.140418,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "zai/glm-4.6": {
                "accuracy": 36.538,
                "latency": 780.014,
                "stderr": 6.743,
                "cost_per_test": 0.196863,
                "temperature": 0.6,
                "top_p": 1.0,
                "max_output_tokens": 122880,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "fireworks/gpt-oss-120b": {
                "accuracy": 28.846,
                "latency": 164.462,
                "stderr": 6.344,
                "cost_per_test": 0.058849,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "mistralai/mistral-large-2512": {
                "accuracy": 19.231,
                "latency": 82.413,
                "stderr": 5.519,
                "cost_per_test": 0.106025,
                "temperature": 0.065,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "openai/gpt-4o-2024-08-06": {
                "accuracy": 5.769,
                "latency": 32.115,
                "stderr": 3.265,
                "cost_per_test": 0.156592,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "cohere/command-a-03-2025": {
                "accuracy": 3.846,
                "latency": 619.74,
                "stderr": 2.693,
                "cost_per_test": 2.610665,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 8000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "fireworks/deepseek-v3p2-thinking": {
                "accuracy": 1.923,
                "latency": 2097.066,
                "stderr": 1.923,
                "cost_per_test": 0.177082,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 20480,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "fireworks/deepseek-v3p2": {
                "accuracy": 0.0,
                "latency": 82.458,
                "stderr": 0.0,
                "cost_per_test": 0.030347,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 20480,
                "reasoning": null,
                "reasoning_effort": "none",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "ai21labs/jamba-large-1.7": {
                "accuracy": 0.0,
                "latency": 2423.531,
                "stderr": 0.0,
                "cost_per_test": 5.212055,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            }
        },
        "financial_modeling": {
            "google/gemini-3-pro-preview": {
                "accuracy": 80.0,
                "latency": 124.416,
                "stderr": 7.428,
                "cost_per_test": 0.460674,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 80.0,
                "latency": 702.413,
                "stderr": 7.428,
                "cost_per_test": 2.459956,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 76.667,
                "latency": 90.361,
                "stderr": 7.854,
                "cost_per_test": 0.043822,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 76.667,
                "latency": 174.422,
                "stderr": 7.854,
                "cost_per_test": 1.290291,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 76.667,
                "latency": 410.378,
                "stderr": 7.854,
                "cost_per_test": 1.160954,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 73.333,
                "latency": 334.791,
                "stderr": 8.212,
                "cost_per_test": 1.150993,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 73.333,
                "latency": 389.223,
                "stderr": 8.212,
                "cost_per_test": 1.436833,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-4-7": {
                "accuracy": 73.333,
                "latency": 435.531,
                "stderr": 8.212,
                "cost_per_test": 1.096073,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 73.333,
                "latency": 552.982,
                "stderr": 8.212,
                "cost_per_test": 1.20004,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 73.333,
                "latency": 1202.49,
                "stderr": 8.212,
                "cost_per_test": 0.499296,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3.6-max-preview": {
                "accuracy": 73.333,
                "latency": 1706.268,
                "stderr": 8.212,
                "cost_per_test": 0.415803,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "meta/muse_spark": {
                "accuracy": 72.414,
                "latency": 378.881,
                "stderr": 8.447,
                "cost_per_test": 0.057961,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 70.0,
                "latency": 183.532,
                "stderr": 8.51,
                "cost_per_test": 0.961966,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 70.0,
                "latency": 250.059,
                "stderr": 8.51,
                "cost_per_test": 0.317882,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 70.0,
                "latency": 253.107,
                "stderr": 8.51,
                "cost_per_test": 0.399324,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 70.0,
                "latency": 482.741,
                "stderr": 8.51,
                "cost_per_test": 0.75367,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-5.1": {
                "accuracy": 70.0,
                "latency": 514.461,
                "stderr": 8.51,
                "cost_per_test": 0.250298,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "grok/grok-4.3": {
                "accuracy": 70.0,
                "latency": 929.162,
                "stderr": 8.51,
                "cost_per_test": 0.420955,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 70.0,
                "latency": 1408.458,
                "stderr": 8.51,
                "cost_per_test": 0.479154,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 66.667,
                "latency": 99.027,
                "stderr": 8.754,
                "cost_per_test": 0.247035,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3.5-flash": {
                "accuracy": 66.667,
                "latency": 250.751,
                "stderr": 8.754,
                "cost_per_test": 0.03387,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 66.667,
                "latency": 313.045,
                "stderr": 8.754,
                "cost_per_test": 0.107404,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "zai/glm-5-thinking": {
                "accuracy": 66.667,
                "latency": 457.723,
                "stderr": 8.754,
                "cost_per_test": 0.365974,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 66.667,
                "latency": 727.946,
                "stderr": 8.754,
                "cost_per_test": 0.74239,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 63.333,
                "latency": 248.957,
                "stderr": 8.949,
                "cost_per_test": 0.140131,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "alibaba/qwen3.5-plus-thinking": {
                "accuracy": 63.333,
                "latency": 342.015,
                "stderr": 8.949,
                "cost_per_test": 0.1964,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "zai/glm-4.7": {
                "accuracy": 63.333,
                "latency": 450.508,
                "stderr": 8.949,
                "cost_per_test": 0.185798,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 63.333,
                "latency": 750.326,
                "stderr": 8.949,
                "cost_per_test": 0.451966,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 63.333,
                "latency": 922.306,
                "stderr": 8.949,
                "cost_per_test": 0.105375,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "google/gemma-4-31b-it": {
                "accuracy": 63.333,
                "latency": 4797.741,
                "stderr": 8.949,
                "cost_per_test": 0.0,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-4-1-fast-non-reasoning": {
                "accuracy": 60.0,
                "latency": 89.739,
                "stderr": 9.097,
                "cost_per_test": 0.090536,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "mistralai/mistral-medium-3.5": {
                "accuracy": 60.0,
                "latency": 319.809,
                "stderr": 9.097,
                "cost_per_test": 0.841381,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 60.0,
                "latency": 566.028,
                "stderr": 9.097,
                "cost_per_test": 0.119282,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 60.0,
                "latency": 1260.423,
                "stderr": 9.097,
                "cost_per_test": 0.342165,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 56.667,
                "latency": 75.244,
                "stderr": 9.202,
                "cost_per_test": 0.065274,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-2.5-pro": {
                "accuracy": 56.667,
                "latency": 80.976,
                "stderr": 9.202,
                "cost_per_test": 0.283662,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3-max": {
                "accuracy": 56.667,
                "latency": 245.89,
                "stderr": 9.202,
                "cost_per_test": 0.567798,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 56.667,
                "latency": 285.823,
                "stderr": 9.202,
                "cost_per_test": 0.049297,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4-0709": {
                "accuracy": 56.667,
                "latency": 616.412,
                "stderr": 9.202,
                "cost_per_test": 1.2388,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "minimax/MiniMax-M2.1": {
                "accuracy": 53.333,
                "latency": 1769.375,
                "stderr": 9.264,
                "cost_per_test": 0.071182,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "kimi/kimi-k2-thinking": {
                "accuracy": 50.0,
                "latency": 651.325,
                "stderr": 9.285,
                "cost_per_test": 0.190709,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "minimax/MiniMax-M2.5": {
                "accuracy": 50.0,
                "latency": 1890.908,
                "stderr": 9.285,
                "cost_per_test": 0.132792,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "grok/grok-4-fast-reasoning": {
                "accuracy": 46.667,
                "latency": 59.246,
                "stderr": 9.264,
                "cost_per_test": 0.042317,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "zai/glm-4.6": {
                "accuracy": 46.667,
                "latency": 341.811,
                "stderr": 9.264,
                "cost_per_test": 0.090662,
                "temperature": 0.6,
                "top_p": 1.0,
                "max_output_tokens": 122880,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "mistralai/mistral-large-2512": {
                "accuracy": 26.667,
                "latency": 59.601,
                "stderr": 8.212,
                "cost_per_test": 0.045147,
                "temperature": 0.065,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "fireworks/gpt-oss-120b": {
                "accuracy": 20.0,
                "latency": 118.776,
                "stderr": 7.428,
                "cost_per_test": 0.031929,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "openai/gpt-4o-2024-08-06": {
                "accuracy": 16.667,
                "latency": 29.683,
                "stderr": 6.92,
                "cost_per_test": 0.102426,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "cohere/command-a-03-2025": {
                "accuracy": 13.333,
                "latency": 310.964,
                "stderr": 6.312,
                "cost_per_test": 1.389055,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 8000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "ai21labs/jamba-large-1.7": {
                "accuracy": 3.333,
                "latency": 2125.152,
                "stderr": 3.333,
                "cost_per_test": 4.678224,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "fireworks/deepseek-v3p2": {
                "accuracy": 0.0,
                "latency": 94.525,
                "stderr": 0.0,
                "cost_per_test": 0.03119,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 20480,
                "reasoning": null,
                "reasoning_effort": "none",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "fireworks/deepseek-v3p2-thinking": {
                "accuracy": 0.0,
                "latency": 1953.044,
                "stderr": 0.0,
                "cost_per_test": 0.242527,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 20480,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            }
        },
        "market_analysis": {
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 33.333,
                "latency": 660.217,
                "stderr": 10.541,
                "cost_per_test": 2.83606,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 30.0,
                "latency": 1455.418,
                "stderr": 10.513,
                "cost_per_test": 8.77053,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-2.5-pro": {
                "accuracy": 28.571,
                "latency": 162.917,
                "stderr": 10.102,
                "cost_per_test": 0.718221,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 28.571,
                "latency": 292.132,
                "stderr": 10.102,
                "cost_per_test": 2.878149,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-4-7": {
                "accuracy": 28.571,
                "latency": 315.746,
                "stderr": 10.102,
                "cost_per_test": 3.144157,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3.5-plus-thinking": {
                "accuracy": 28.571,
                "latency": 527.017,
                "stderr": 10.102,
                "cost_per_test": 0.430327,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 28.571,
                "latency": 1084.441,
                "stderr": 10.102,
                "cost_per_test": 1.929355,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3.6-max-preview": {
                "accuracy": 28.571,
                "latency": 1751.384,
                "stderr": 10.102,
                "cost_per_test": 1.005347,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "google/gemma-4-31b-it": {
                "accuracy": 28.571,
                "latency": 6662.098,
                "stderr": 10.102,
                "cost_per_test": 0.0,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 23.81,
                "latency": 134.762,
                "stderr": 9.524,
                "cost_per_test": 0.100889,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 23.81,
                "latency": 347.785,
                "stderr": 9.524,
                "cost_per_test": 2.348871,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-4-0709": {
                "accuracy": 23.81,
                "latency": 351.378,
                "stderr": 9.524,
                "cost_per_test": 1.410206,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "alibaba/qwen3-max": {
                "accuracy": 23.81,
                "latency": 405.833,
                "stderr": 9.524,
                "cost_per_test": 1.243792,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 23.81,
                "latency": 484.999,
                "stderr": 9.524,
                "cost_per_test": 0.296305,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 23.81,
                "latency": 668.313,
                "stderr": 9.524,
                "cost_per_test": 1.004585,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "mistralai/mistral-medium-3.5": {
                "accuracy": 23.81,
                "latency": 926.621,
                "stderr": 9.524,
                "cost_per_test": 1.773243,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "zai/glm-4.6": {
                "accuracy": 23.81,
                "latency": 968.714,
                "stderr": 9.524,
                "cost_per_test": 0.259727,
                "temperature": 0.6,
                "top_p": 1.0,
                "max_output_tokens": 122880,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 23.81,
                "latency": 1147.774,
                "stderr": 9.524,
                "cost_per_test": 0.289282,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 23.81,
                "latency": 1287.652,
                "stderr": 9.524,
                "cost_per_test": 3.410783,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 23.81,
                "latency": 2121.515,
                "stderr": 9.524,
                "cost_per_test": 0.882786,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "meta/muse_spark": {
                "accuracy": 20.0,
                "latency": 489.243,
                "stderr": 9.177,
                "cost_per_test": 0.105514,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "grok/grok-4-fast-reasoning": {
                "accuracy": 19.048,
                "latency": 85.042,
                "stderr": 8.781,
                "cost_per_test": 0.079972,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 19.048,
                "latency": 184.376,
                "stderr": 8.781,
                "cost_per_test": 0.682368,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 19.048,
                "latency": 195.397,
                "stderr": 8.781,
                "cost_per_test": 0.56879,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 19.048,
                "latency": 279.606,
                "stderr": 8.781,
                "cost_per_test": 1.335837,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3.5-flash": {
                "accuracy": 19.048,
                "latency": 329.225,
                "stderr": 8.781,
                "cost_per_test": 0.073541,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 19.048,
                "latency": 426.006,
                "stderr": 8.781,
                "cost_per_test": 0.31913,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 19.048,
                "latency": 462.607,
                "stderr": 8.781,
                "cost_per_test": 0.83318,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 19.048,
                "latency": 486.518,
                "stderr": 8.781,
                "cost_per_test": 2.049798,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-4.3": {
                "accuracy": 19.048,
                "latency": 810.12,
                "stderr": 8.781,
                "cost_per_test": 0.611053,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "zai/glm-4.7": {
                "accuracy": 19.048,
                "latency": 851.132,
                "stderr": 8.781,
                "cost_per_test": 0.358054,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 19.048,
                "latency": 1227.559,
                "stderr": 8.781,
                "cost_per_test": 1.242821,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 19.048,
                "latency": 1324.502,
                "stderr": 8.781,
                "cost_per_test": 0.906104,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-5.1": {
                "accuracy": 19.048,
                "latency": 1692.593,
                "stderr": 8.781,
                "cost_per_test": 0.620063,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 19.048,
                "latency": 2267.801,
                "stderr": 8.781,
                "cost_per_test": 0.45325,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 14.286,
                "latency": 323.539,
                "stderr": 7.825,
                "cost_per_test": 0.049297,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 14.286,
                "latency": 512.553,
                "stderr": 7.825,
                "cost_per_test": 0.943409,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-5-thinking": {
                "accuracy": 14.286,
                "latency": 878.316,
                "stderr": 7.825,
                "cost_per_test": 0.768093,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 9.524,
                "latency": 42.368,
                "stderr": 6.564,
                "cost_per_test": 0.086276,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-4-1-fast-non-reasoning": {
                "accuracy": 9.524,
                "latency": 115.252,
                "stderr": 6.564,
                "cost_per_test": 0.150176,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "mistralai/mistral-large-2512": {
                "accuracy": 9.524,
                "latency": 242.192,
                "stderr": 6.564,
                "cost_per_test": 0.167282,
                "temperature": 0.065,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "kimi/kimi-k2-thinking": {
                "accuracy": 9.524,
                "latency": 840.887,
                "stderr": 6.564,
                "cost_per_test": 0.325672,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 9.524,
                "latency": 919.996,
                "stderr": 6.564,
                "cost_per_test": 0.251806,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "minimax/MiniMax-M2.1": {
                "accuracy": 9.524,
                "latency": 2076.457,
                "stderr": 6.564,
                "cost_per_test": 0.201614,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "openai/gpt-4o-2024-08-06": {
                "accuracy": 4.762,
                "latency": 39.144,
                "stderr": 4.762,
                "cost_per_test": 0.218676,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "fireworks/deepseek-v3p2": {
                "accuracy": 0.0,
                "latency": 100.176,
                "stderr": 0.0,
                "cost_per_test": 0.035936,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 20480,
                "reasoning": null,
                "reasoning_effort": "none",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "fireworks/gpt-oss-120b": {
                "accuracy": 0.0,
                "latency": 322.302,
                "stderr": 0.0,
                "cost_per_test": 0.113881,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "cohere/command-a-03-2025": {
                "accuracy": 0.0,
                "latency": 828.177,
                "stderr": 0.0,
                "cost_per_test": 3.657835,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 8000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "fireworks/deepseek-v3p2-thinking": {
                "accuracy": 0.0,
                "latency": 1806.978,
                "stderr": 0.0,
                "cost_per_test": 0.191698,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 20480,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "ai21labs/jamba-large-1.7": {
                "accuracy": 0.0,
                "latency": 2384.528,
                "stderr": 0.0,
                "cost_per_test": 5.254214,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "minimax/MiniMax-M2.5": {
                "accuracy": 0.0,
                "latency": 2385.853,
                "stderr": 0.0,
                "cost_per_test": 0.292046,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            }
        },
        "beat_or_miss": {
            "meta/muse_spark": {
                "accuracy": 58.14,
                "latency": 481.812,
                "stderr": 7.612,
                "cost_per_test": 0.085132,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "anthropic/claude-opus-4-7": {
                "accuracy": 55.814,
                "latency": 374.107,
                "stderr": 7.663,
                "cost_per_test": 0.599903,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 53.488,
                "latency": 227.592,
                "stderr": 7.696,
                "cost_per_test": 0.806863,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 53.488,
                "latency": 313.298,
                "stderr": 7.696,
                "cost_per_test": 0.476863,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 53.488,
                "latency": 694.343,
                "stderr": 7.696,
                "cost_per_test": 0.932321,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 51.163,
                "latency": 408.819,
                "stderr": 7.713,
                "cost_per_test": 1.08652,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 51.163,
                "latency": 1030.721,
                "stderr": 7.713,
                "cost_per_test": 0.441112,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 48.837,
                "latency": 466.554,
                "stderr": 7.713,
                "cost_per_test": 0.394886,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 48.837,
                "latency": 805.085,
                "stderr": 7.713,
                "cost_per_test": 2.331121,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 46.512,
                "latency": 315.296,
                "stderr": 7.696,
                "cost_per_test": 0.830493,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "zai/glm-5-thinking": {
                "accuracy": 46.512,
                "latency": 432.167,
                "stderr": 7.696,
                "cost_per_test": 0.264102,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "grok/grok-4.3": {
                "accuracy": 46.512,
                "latency": 707.234,
                "stderr": 7.696,
                "cost_per_test": 0.283625,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 44.186,
                "latency": 52.72,
                "stderr": 7.663,
                "cost_per_test": 0.053675,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 44.186,
                "latency": 237.492,
                "stderr": 7.663,
                "cost_per_test": 0.541133,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-4-0709": {
                "accuracy": 44.186,
                "latency": 302.839,
                "stderr": 7.663,
                "cost_per_test": 0.759328,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 44.186,
                "latency": 313.958,
                "stderr": 7.663,
                "cost_per_test": 0.330609,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 44.186,
                "latency": 318.029,
                "stderr": 7.663,
                "cost_per_test": 0.373814,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 44.186,
                "latency": 372.445,
                "stderr": 7.663,
                "cost_per_test": 0.289319,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "mistralai/mistral-medium-3.5": {
                "accuracy": 44.186,
                "latency": 405.106,
                "stderr": 7.663,
                "cost_per_test": 1.361335,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 44.186,
                "latency": 550.367,
                "stderr": 7.663,
                "cost_per_test": 0.710299,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 44.186,
                "latency": 694.419,
                "stderr": 7.663,
                "cost_per_test": 0.14744,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 41.86,
                "latency": 109.975,
                "stderr": 7.612,
                "cost_per_test": 0.031817,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "zai/glm-5.1": {
                "accuracy": 41.86,
                "latency": 392.23,
                "stderr": 7.612,
                "cost_per_test": 0.22337,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 41.86,
                "latency": 794.706,
                "stderr": 7.612,
                "cost_per_test": 0.427528,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 41.86,
                "latency": 1154.022,
                "stderr": 7.612,
                "cost_per_test": 0.213493,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 41.86,
                "latency": 1919.684,
                "stderr": 7.612,
                "cost_per_test": 0.367022,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 39.535,
                "latency": 279.761,
                "stderr": 7.544,
                "cost_per_test": 0.143743,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "alibaba/qwen3.6-max-preview": {
                "accuracy": 39.535,
                "latency": 1798.696,
                "stderr": 7.544,
                "cost_per_test": 0.766829,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "google/gemma-4-31b-it": {
                "accuracy": 39.535,
                "latency": 5122.795,
                "stderr": 7.544,
                "cost_per_test": 0.0,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 37.209,
                "latency": 193.211,
                "stderr": 7.458,
                "cost_per_test": 0.219043,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 37.209,
                "latency": 1550.714,
                "stderr": 7.458,
                "cost_per_test": 0.318247,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-4-fast-reasoning": {
                "accuracy": 34.884,
                "latency": 62.306,
                "stderr": 7.354,
                "cost_per_test": 0.037087,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 34.884,
                "latency": 289.09,
                "stderr": 7.354,
                "cost_per_test": 0.049297,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-4.7": {
                "accuracy": 34.884,
                "latency": 615.662,
                "stderr": 7.354,
                "cost_per_test": 0.194707,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 32.558,
                "latency": 140.881,
                "stderr": 7.231,
                "cost_per_test": 0.481883,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3.5-flash": {
                "accuracy": 32.558,
                "latency": 255.648,
                "stderr": 7.231,
                "cost_per_test": 0.073309,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "alibaba/qwen3.5-plus-thinking": {
                "accuracy": 32.558,
                "latency": 448.188,
                "stderr": 7.231,
                "cost_per_test": 0.318701,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "minimax/MiniMax-M2.1": {
                "accuracy": 32.558,
                "latency": 1367.969,
                "stderr": 7.231,
                "cost_per_test": 0.135531,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "alibaba/qwen3-max": {
                "accuracy": 30.233,
                "latency": 364.396,
                "stderr": 7.087,
                "cost_per_test": 0.503056,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "minimax/MiniMax-M2.5": {
                "accuracy": 27.907,
                "latency": 1978.035,
                "stderr": 6.921,
                "cost_per_test": 0.185154,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "grok/grok-4-1-fast-non-reasoning": {
                "accuracy": 25.581,
                "latency": 125.418,
                "stderr": 6.733,
                "cost_per_test": 0.043571,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-2.5-pro": {
                "accuracy": 18.605,
                "latency": 112.475,
                "stderr": 6.005,
                "cost_per_test": 0.243167,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "zai/glm-4.6": {
                "accuracy": 16.279,
                "latency": 309.916,
                "stderr": 5.696,
                "cost_per_test": 0.0895,
                "temperature": 0.6,
                "top_p": 1.0,
                "max_output_tokens": 122880,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "kimi/kimi-k2-thinking": {
                "accuracy": 16.279,
                "latency": 811.925,
                "stderr": 5.696,
                "cost_per_test": 0.26498,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "fireworks/gpt-oss-120b": {
                "accuracy": 11.628,
                "latency": 219.987,
                "stderr": 4.946,
                "cost_per_test": 0.09845,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "mistralai/mistral-large-2512": {
                "accuracy": 9.302,
                "latency": 76.832,
                "stderr": 4.482,
                "cost_per_test": 0.056897,
                "temperature": 0.065,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "openai/gpt-4o-2024-08-06": {
                "accuracy": 2.326,
                "latency": 33.093,
                "stderr": 2.326,
                "cost_per_test": 0.144942,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "fireworks/deepseek-v3p2-thinking": {
                "accuracy": 2.326,
                "latency": 1723.609,
                "stderr": 2.326,
                "cost_per_test": 0.164577,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 20480,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "fireworks/deepseek-v3p2": {
                "accuracy": 0.0,
                "latency": 121.675,
                "stderr": 0.0,
                "cost_per_test": 0.053694,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 20480,
                "reasoning": null,
                "reasoning_effort": "none",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "cohere/command-a-03-2025": {
                "accuracy": 0.0,
                "latency": 645.867,
                "stderr": 0.0,
                "cost_per_test": 2.662162,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 8000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "ai21labs/jamba-large-1.7": {
                "accuracy": 0.0,
                "latency": 2568.085,
                "stderr": 0.0,
                "cost_per_test": 5.704158,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            }
        },
        "trends": {
            "zai/glm-5.1": {
                "accuracy": 71.429,
                "latency": 551.123,
                "stderr": 10.102,
                "cost_per_test": 0.626575,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 61.905,
                "latency": 375.267,
                "stderr": 10.859,
                "cost_per_test": 2.24563,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 61.905,
                "latency": 469.602,
                "stderr": 10.859,
                "cost_per_test": 2.572976,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 61.905,
                "latency": 748.342,
                "stderr": 10.859,
                "cost_per_test": 1.146575,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "zai/glm-5-thinking": {
                "accuracy": 61.905,
                "latency": 828.379,
                "stderr": 10.859,
                "cost_per_test": 0.888307,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 57.143,
                "latency": 375.885,
                "stderr": 11.066,
                "cost_per_test": 1.355841,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-4-7": {
                "accuracy": 57.143,
                "latency": 407.491,
                "stderr": 11.066,
                "cost_per_test": 3.225803,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "meta/muse_spark": {
                "accuracy": 57.143,
                "latency": 452.771,
                "stderr": 11.066,
                "cost_per_test": 0.094787,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "grok/grok-4-0709": {
                "accuracy": 57.143,
                "latency": 455.685,
                "stderr": 11.066,
                "cost_per_test": 2.147365,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 57.143,
                "latency": 1304.305,
                "stderr": 11.066,
                "cost_per_test": 8.949381,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 52.381,
                "latency": 123.989,
                "stderr": 11.168,
                "cost_per_test": 0.134948,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 52.381,
                "latency": 285.475,
                "stderr": 11.168,
                "cost_per_test": 3.499236,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 52.381,
                "latency": 860.781,
                "stderr": 11.168,
                "cost_per_test": 3.760124,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 52.381,
                "latency": 1103.063,
                "stderr": 11.168,
                "cost_per_test": 0.804416,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 52.381,
                "latency": 2121.557,
                "stderr": 11.168,
                "cost_per_test": 1.020273,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 52.381,
                "latency": 2861.248,
                "stderr": 11.168,
                "cost_per_test": 1.104933,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-2.5-pro": {
                "accuracy": 52.381,
                "latency": 13244.666,
                "stderr": 11.168,
                "cost_per_test": 0.834324,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 47.619,
                "latency": 287.285,
                "stderr": 11.168,
                "cost_per_test": 3.018569,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 47.619,
                "latency": 366.26,
                "stderr": 11.168,
                "cost_per_test": 0.121103,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "alibaba/qwen3.5-plus-thinking": {
                "accuracy": 47.619,
                "latency": 457.489,
                "stderr": 11.168,
                "cost_per_test": 0.360102,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 47.619,
                "latency": 1039.008,
                "stderr": 11.168,
                "cost_per_test": 0.499387,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 47.619,
                "latency": 1063.045,
                "stderr": 11.168,
                "cost_per_test": 2.47147,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 42.857,
                "latency": 161.954,
                "stderr": 11.066,
                "cost_per_test": 1.068203,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 42.857,
                "latency": 229.413,
                "stderr": 11.066,
                "cost_per_test": 0.685992,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 42.857,
                "latency": 358.274,
                "stderr": 11.066,
                "cost_per_test": 0.049297,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 42.857,
                "latency": 496.976,
                "stderr": 11.066,
                "cost_per_test": 0.175189,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 42.857,
                "latency": 521.664,
                "stderr": 11.066,
                "cost_per_test": 1.048848,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 42.857,
                "latency": 614.988,
                "stderr": 11.066,
                "cost_per_test": 0.410589,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "grok/grok-4.3": {
                "accuracy": 42.857,
                "latency": 922.242,
                "stderr": 11.066,
                "cost_per_test": 0.882921,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "alibaba/qwen3.6-max-preview": {
                "accuracy": 42.857,
                "latency": 1832.983,
                "stderr": 11.066,
                "cost_per_test": 1.018647,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "google/gemma-4-31b-it": {
                "accuracy": 42.857,
                "latency": 7514.167,
                "stderr": 11.066,
                "cost_per_test": 0.0,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-4-1-fast-non-reasoning": {
                "accuracy": 38.095,
                "latency": 101.457,
                "stderr": 10.859,
                "cost_per_test": 0.209225,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 38.095,
                "latency": 255.154,
                "stderr": 10.859,
                "cost_per_test": 0.835799,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "kimi/kimi-k2-thinking": {
                "accuracy": 38.095,
                "latency": 563.303,
                "stderr": 10.859,
                "cost_per_test": 0.316354,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 38.095,
                "latency": 915.831,
                "stderr": 10.859,
                "cost_per_test": 0.228751,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-4.7": {
                "accuracy": 38.095,
                "latency": 1011.628,
                "stderr": 10.859,
                "cost_per_test": 0.465608,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 38.095,
                "latency": 2035.952,
                "stderr": 10.859,
                "cost_per_test": 0.212656,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "alibaba/qwen3.5-flash": {
                "accuracy": 33.333,
                "latency": 287.679,
                "stderr": 10.541,
                "cost_per_test": 0.073875,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "mistralai/mistral-medium-3.5": {
                "accuracy": 33.333,
                "latency": 1518.904,
                "stderr": 10.541,
                "cost_per_test": 2.208876,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "zai/glm-4.6": {
                "accuracy": 33.333,
                "latency": 2294.777,
                "stderr": 10.541,
                "cost_per_test": 0.662894,
                "temperature": 0.6,
                "top_p": 1.0,
                "max_output_tokens": 122880,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "grok/grok-4-fast-reasoning": {
                "accuracy": 28.571,
                "latency": 94.881,
                "stderr": 10.102,
                "cost_per_test": 0.078802,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "alibaba/qwen3-max": {
                "accuracy": 28.571,
                "latency": 744.421,
                "stderr": 10.102,
                "cost_per_test": 2.273385,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "minimax/MiniMax-M2.5": {
                "accuracy": 28.571,
                "latency": 1558.012,
                "stderr": 10.102,
                "cost_per_test": 0.276367,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "minimax/MiniMax-M2.1": {
                "accuracy": 19.048,
                "latency": 887.558,
                "stderr": 8.781,
                "cost_per_test": 0.153919,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "fireworks/gpt-oss-120b": {
                "accuracy": 14.286,
                "latency": 332.137,
                "stderr": 7.825,
                "cost_per_test": 0.123503,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "mistralai/mistral-large-2512": {
                "accuracy": 9.524,
                "latency": 186.302,
                "stderr": 6.564,
                "cost_per_test": 0.220094,
                "temperature": 0.065,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "openai/gpt-4o-2024-08-06": {
                "accuracy": 0.0,
                "latency": 51.374,
                "stderr": 0.0,
                "cost_per_test": 0.334875,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "fireworks/deepseek-v3p2": {
                "accuracy": 0.0,
                "latency": 148.864,
                "stderr": 0.0,
                "cost_per_test": 0.090419,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 20480,
                "reasoning": null,
                "reasoning_effort": "none",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "fireworks/deepseek-v3p2-thinking": {
                "accuracy": 0.0,
                "latency": 923.166,
                "stderr": 0.0,
                "cost_per_test": 0.107123,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 20480,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "cohere/command-a-03-2025": {
                "accuracy": 0.0,
                "latency": 1737.232,
                "stderr": 0.0,
                "cost_per_test": 5.766424,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 8000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "ai21labs/jamba-large-1.7": {
                "accuracy": 0.0,
                "latency": 3386.613,
                "stderr": 0.0,
                "cost_per_test": 6.223644,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            }
        },
        "adjustments": {
            "anthropic/claude-opus-4-7": {
                "accuracy": 74.074,
                "latency": 182.077,
                "stderr": 8.594,
                "cost_per_test": 1.191471,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 66.667,
                "latency": 489.008,
                "stderr": 9.245,
                "cost_per_test": 1.186809,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 66.667,
                "latency": 505.643,
                "stderr": 9.245,
                "cost_per_test": 1.628087,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 66.667,
                "latency": 722.318,
                "stderr": 9.245,
                "cost_per_test": 0.789724,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 62.963,
                "latency": 131.733,
                "stderr": 9.471,
                "cost_per_test": 0.570259,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 62.963,
                "latency": 174.17,
                "stderr": 9.471,
                "cost_per_test": 1.721915,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-4-0709": {
                "accuracy": 62.963,
                "latency": 272.785,
                "stderr": 9.471,
                "cost_per_test": 1.118074,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 62.963,
                "latency": 343.905,
                "stderr": 9.471,
                "cost_per_test": 0.493588,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "meta/muse_spark": {
                "accuracy": 62.963,
                "latency": 569.606,
                "stderr": 9.471,
                "cost_per_test": 0.064759,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 59.259,
                "latency": 287.965,
                "stderr": 9.636,
                "cost_per_test": 1.284942,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3.5-plus-thinking": {
                "accuracy": 59.259,
                "latency": 345.466,
                "stderr": 9.636,
                "cost_per_test": 0.194466,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 59.259,
                "latency": 974.418,
                "stderr": 9.636,
                "cost_per_test": 3.733275,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemma-4-31b-it": {
                "accuracy": 59.259,
                "latency": 5687.288,
                "stderr": 9.636,
                "cost_per_test": 0.0,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-4-fast-reasoning": {
                "accuracy": 55.556,
                "latency": 64.864,
                "stderr": 9.745,
                "cost_per_test": 0.048333,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 55.556,
                "latency": 183.568,
                "stderr": 9.745,
                "cost_per_test": 1.272621,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 55.556,
                "latency": 334.922,
                "stderr": 9.745,
                "cost_per_test": 0.122567,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "zai/glm-5.1": {
                "accuracy": 55.556,
                "latency": 431.447,
                "stderr": 9.745,
                "cost_per_test": 0.335755,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 55.556,
                "latency": 611.014,
                "stderr": 9.745,
                "cost_per_test": 0.94679,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4.3": {
                "accuracy": 55.556,
                "latency": 884.886,
                "stderr": 9.745,
                "cost_per_test": 0.4552,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 55.556,
                "latency": 1637.988,
                "stderr": 9.745,
                "cost_per_test": 0.581707,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 51.852,
                "latency": 49.832,
                "stderr": 9.799,
                "cost_per_test": 0.380617,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 51.852,
                "latency": 92.425,
                "stderr": 9.799,
                "cost_per_test": 0.061964,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "alibaba/qwen3-max": {
                "accuracy": 51.852,
                "latency": 264.269,
                "stderr": 9.799,
                "cost_per_test": 0.724925,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 51.852,
                "latency": 289.388,
                "stderr": 9.799,
                "cost_per_test": 0.184042,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 51.852,
                "latency": 810.693,
                "stderr": 9.799,
                "cost_per_test": 1.231364,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 51.852,
                "latency": 1351.6,
                "stderr": 9.799,
                "cost_per_test": 0.555615,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 48.148,
                "latency": 49.92,
                "stderr": 9.799,
                "cost_per_test": 0.066932,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 48.148,
                "latency": 115.711,
                "stderr": 9.799,
                "cost_per_test": 0.349876,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 48.148,
                "latency": 265.423,
                "stderr": 9.799,
                "cost_per_test": 0.049297,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 48.148,
                "latency": 595.795,
                "stderr": 9.799,
                "cost_per_test": 0.128702,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-5-thinking": {
                "accuracy": 48.148,
                "latency": 893.72,
                "stderr": 9.799,
                "cost_per_test": 0.817842,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 48.148,
                "latency": 1004.029,
                "stderr": 9.799,
                "cost_per_test": 0.564512,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3.6-max-preview": {
                "accuracy": 48.148,
                "latency": 1876.653,
                "stderr": 9.799,
                "cost_per_test": 0.638055,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "grok/grok-4-1-fast-non-reasoning": {
                "accuracy": 44.444,
                "latency": 115.364,
                "stderr": 9.745,
                "cost_per_test": 0.08066,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 44.444,
                "latency": 920.916,
                "stderr": 9.745,
                "cost_per_test": 0.430723,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 44.444,
                "latency": 1321.899,
                "stderr": 9.745,
                "cost_per_test": 0.171421,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "minimax/MiniMax-M2.5": {
                "accuracy": 44.444,
                "latency": 1555.742,
                "stderr": 9.745,
                "cost_per_test": 0.192613,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "mistralai/mistral-medium-3.5": {
                "accuracy": 40.741,
                "latency": 549.572,
                "stderr": 9.636,
                "cost_per_test": 0.955774,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "zai/glm-4.7": {
                "accuracy": 40.741,
                "latency": 675.345,
                "stderr": 9.636,
                "cost_per_test": 0.289165,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "kimi/kimi-k2-thinking": {
                "accuracy": 37.037,
                "latency": 860.148,
                "stderr": 9.471,
                "cost_per_test": 0.291967,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "alibaba/qwen3.5-flash": {
                "accuracy": 33.333,
                "latency": 246.675,
                "stderr": 9.245,
                "cost_per_test": 0.052102,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "google/gemini-2.5-pro": {
                "accuracy": 29.63,
                "latency": 69.667,
                "stderr": 8.955,
                "cost_per_test": 0.279162,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "zai/glm-4.6": {
                "accuracy": 29.63,
                "latency": 538.672,
                "stderr": 8.955,
                "cost_per_test": 0.17553,
                "temperature": 0.6,
                "top_p": 1.0,
                "max_output_tokens": 122880,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "mistralai/mistral-large-2512": {
                "accuracy": 25.926,
                "latency": 86.946,
                "stderr": 8.594,
                "cost_per_test": 0.07357,
                "temperature": 0.065,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "minimax/MiniMax-M2.1": {
                "accuracy": 25.926,
                "latency": 2080.543,
                "stderr": 8.594,
                "cost_per_test": 0.127447,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "fireworks/gpt-oss-120b": {
                "accuracy": 22.222,
                "latency": 194.066,
                "stderr": 8.153,
                "cost_per_test": 0.061203,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "fireworks/deepseek-v3p2-thinking": {
                "accuracy": 7.407,
                "latency": 3357.102,
                "stderr": 5.136,
                "cost_per_test": 0.350087,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 20480,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "openai/gpt-4o-2024-08-06": {
                "accuracy": 0.0,
                "latency": 42.681,
                "stderr": 0.0,
                "cost_per_test": 0.151455,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "fireworks/deepseek-v3p2": {
                "accuracy": 0.0,
                "latency": 94.998,
                "stderr": 0.0,
                "cost_per_test": 0.028208,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 20480,
                "reasoning": null,
                "reasoning_effort": "none",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "cohere/command-a-03-2025": {
                "accuracy": 0.0,
                "latency": 371.731,
                "stderr": 0.0,
                "cost_per_test": 1.776897,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 8000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "ai21labs/jamba-large-1.7": {
                "accuracy": 0.0,
                "latency": 2432.931,
                "stderr": 0.0,
                "cost_per_test": 5.355582,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            }
        },
        "vals_index_subset": {
            "anthropic/claude-opus-4-7": {
                "accuracy": 72.222,
                "latency": 273.714,
                "stderr": 4.748,
                "cost_per_test": 0.785991,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 68.889,
                "latency": 347.536,
                "stderr": 4.907,
                "cost_per_test": 1.334413,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "meta/muse_spark": {
                "accuracy": 65.909,
                "latency": 475.798,
                "stderr": 5.082,
                "cost_per_test": 0.064385,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 64.444,
                "latency": 312.428,
                "stderr": 5.074,
                "cost_per_test": 1.119515,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 64.444,
                "latency": 653.19,
                "stderr": 5.074,
                "cost_per_test": 0.60915,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "zai/glm-5.1": {
                "accuracy": 64.444,
                "latency": 715.061,
                "stderr": 5.074,
                "cost_per_test": 0.317186,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 63.333,
                "latency": 593.047,
                "stderr": 5.108,
                "cost_per_test": 0.99647,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 63.333,
                "latency": 899.814,
                "stderr": 5.108,
                "cost_per_test": 1.36087,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 62.222,
                "latency": 210.976,
                "stderr": 5.139,
                "cost_per_test": 1.115022,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 62.222,
                "latency": 319.034,
                "stderr": 5.139,
                "cost_per_test": 0.15739,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 62.222,
                "latency": 1579.835,
                "stderr": 5.139,
                "cost_per_test": 0.506488,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "zai/glm-5-thinking": {
                "accuracy": 61.111,
                "latency": 601.605,
                "stderr": 5.167,
                "cost_per_test": 0.543253,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 60.0,
                "latency": 190.087,
                "stderr": 5.193,
                "cost_per_test": 1.531755,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 60.0,
                "latency": 304.819,
                "stderr": 5.193,
                "cost_per_test": 0.997529,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3.5-plus-thinking": {
                "accuracy": 60.0,
                "latency": 360.728,
                "stderr": 5.193,
                "cost_per_test": 0.236017,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 60.0,
                "latency": 742.706,
                "stderr": 5.193,
                "cost_per_test": 1.555334,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 60.0,
                "latency": 927.737,
                "stderr": 5.193,
                "cost_per_test": 0.517677,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 58.889,
                "latency": 150.386,
                "stderr": 5.216,
                "cost_per_test": 0.610614,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 58.889,
                "latency": 569.69,
                "stderr": 5.216,
                "cost_per_test": 0.483019,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 57.778,
                "latency": 265.478,
                "stderr": 5.235,
                "cost_per_test": 0.174563,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "grok/grok-4-0709": {
                "accuracy": 57.778,
                "latency": 378.378,
                "stderr": 5.235,
                "cost_per_test": 1.094618,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "grok/grok-4.3": {
                "accuracy": 57.778,
                "latency": 772.276,
                "stderr": 5.235,
                "cost_per_test": 0.427426,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 56.667,
                "latency": 77.772,
                "stderr": 5.253,
                "cost_per_test": 0.404538,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 56.667,
                "latency": 1264.086,
                "stderr": 5.253,
                "cost_per_test": 0.164241,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 55.556,
                "latency": 98.738,
                "stderr": 5.267,
                "cost_per_test": 0.064312,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 55.556,
                "latency": 855.084,
                "stderr": 5.267,
                "cost_per_test": 0.343008,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3.6-max-preview": {
                "accuracy": 55.556,
                "latency": 1818.951,
                "stderr": 5.267,
                "cost_per_test": 0.577571,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 54.444,
                "latency": 113.952,
                "stderr": 5.279,
                "cost_per_test": 0.364612,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 53.333,
                "latency": 1047.445,
                "stderr": 5.288,
                "cost_per_test": 0.507004,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 52.222,
                "latency": 640.481,
                "stderr": 5.295,
                "cost_per_test": 0.150327,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "mistralai/mistral-medium-3.5": {
                "accuracy": 51.111,
                "latency": 586.51,
                "stderr": 5.299,
                "cost_per_test": 1.114774,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "zai/glm-4.7": {
                "accuracy": 51.111,
                "latency": 589.974,
                "stderr": 5.299,
                "cost_per_test": 0.237386,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 50.0,
                "latency": 282.476,
                "stderr": 5.3,
                "cost_per_test": 0.01223,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemma-4-31b-it": {
                "accuracy": 50.0,
                "latency": 5581.153,
                "stderr": 5.3,
                "cost_per_test": 0.0,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-4-fast-reasoning": {
                "accuracy": 48.889,
                "latency": 60.939,
                "stderr": 5.299,
                "cost_per_test": 0.049119,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "grok/grok-4-1-fast-non-reasoning": {
                "accuracy": 48.889,
                "latency": 88.056,
                "stderr": 5.299,
                "cost_per_test": 0.097794,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 48.889,
                "latency": 122.395,
                "stderr": 5.299,
                "cost_per_test": 0.072996,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "zai/glm-4.6": {
                "accuracy": 47.778,
                "latency": 459.49,
                "stderr": 5.295,
                "cost_per_test": 0.138663,
                "temperature": 0.6,
                "top_p": 1.0,
                "max_output_tokens": 122880,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "alibaba/qwen3.5-flash": {
                "accuracy": 46.667,
                "latency": 280.999,
                "stderr": 5.288,
                "cost_per_test": 0.058942,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "minimax/MiniMax-M2.5": {
                "accuracy": 45.556,
                "latency": 1415.928,
                "stderr": 5.279,
                "cost_per_test": 0.179788,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "alibaba/qwen3-max": {
                "accuracy": 44.444,
                "latency": 315.78,
                "stderr": 5.267,
                "cost_per_test": 0.722023,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "google/gemini-2.5-pro": {
                "accuracy": 43.333,
                "latency": 1834.194,
                "stderr": 5.253,
                "cost_per_test": 0.324011,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "kimi/kimi-k2-thinking": {
                "accuracy": 38.889,
                "latency": 631.816,
                "stderr": 5.167,
                "cost_per_test": 0.221318,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "minimax/MiniMax-M2.1": {
                "accuracy": 37.778,
                "latency": 2115.919,
                "stderr": 5.139,
                "cost_per_test": 0.124991,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "fireworks/gpt-oss-120b": {
                "accuracy": 23.333,
                "latency": 168.529,
                "stderr": 4.483,
                "cost_per_test": 0.062629,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "mistralai/mistral-large-2512": {
                "accuracy": 20.0,
                "latency": 96.392,
                "stderr": 4.24,
                "cost_per_test": 0.089931,
                "temperature": 0.065,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "openai/gpt-4o-2024-08-06": {
                "accuracy": 12.222,
                "latency": 32.105,
                "stderr": 3.472,
                "cost_per_test": 0.181653,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "cohere/command-a-03-2025": {
                "accuracy": 5.556,
                "latency": 455.697,
                "stderr": 2.428,
                "cost_per_test": 2.106484,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 8000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "fireworks/deepseek-v3p2-thinking": {
                "accuracy": 3.333,
                "latency": 2406.288,
                "stderr": 1.903,
                "cost_per_test": 0.229225,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 20480,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "fireworks/deepseek-v3p2": {
                "accuracy": 0.0,
                "latency": 120.783,
                "stderr": 0.0,
                "cost_per_test": 0.055382,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 20480,
                "reasoning": null,
                "reasoning_effort": "none",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "ai21labs/jamba-large-1.7": {
                "accuracy": 0.0,
                "latency": 2304.029,
                "stderr": 0.0,
                "cost_per_test": 5.01965,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            }
        }
    }
}