{
    "metadata": {
        "benchmark": "TaxEval v2",
        "slug": "tax_eval_v2",
        "description": "A Vals-created set of questions and responses to tax questions",
        "benchmark_id": "tax_eval_v2",
        "family": "tax_eval",
        "version": "2",
        "updated": "2026-09-01",
        "dataset_type": "private",
        "industry": "finance",
        "tasks": {
            "overall": "Overall",
            "correctness": "Correctness",
            "stepwise_reasoning": "Stepwise Reasoning"
        },
        "models": [
            "ai21labs/jamba-1.5-large",
            "ai21labs/jamba-1.5-mini",
            "ai21labs/jamba-large-1.6",
            "ai21labs/jamba-mini-1.6",
            "alibaba/qwen3-max",
            "alibaba/qwen3-max-preview",
            "alibaba/qwen3.5-flash",
            "alibaba/qwen3.6-27b",
            "alibaba/qwen3.6-plus",
            "alibaba/qwen3.7-max",
            "alibaba/qwen3.8-27b",
            "alibaba/qwen3.8-max",
            "ant/ling-3.0-flash-2607",
            "anthropic/claude-3-5-haiku-20241022",
            "anthropic/claude-3-5-sonnet-20241022",
            "anthropic/claude-3-7-sonnet-20250219",
            "anthropic/claude-3-7-sonnet-20250219-thinking",
            "anthropic/claude-fable-5",
            "anthropic/claude-fable-5-1",
            "anthropic/claude-haiku-4-5-20251001-thinking",
            "anthropic/claude-opus-4-1-20250805",
            "anthropic/claude-opus-4-1-20250805-thinking",
            "anthropic/claude-opus-4-20250514",
            "anthropic/claude-opus-4-5-20251101",
            "anthropic/claude-opus-4-5-20251101-thinking",
            "anthropic/claude-opus-4-6-thinking",
            "anthropic/claude-opus-4-7",
            "anthropic/claude-opus-4-8",
            "anthropic/claude-opus-5",
            "anthropic/claude-sonnet-4-20250514",
            "anthropic/claude-sonnet-4-20250514-thinking",
            "anthropic/claude-sonnet-4-5-20250929-thinking",
            "anthropic/claude-sonnet-4-6",
            "anthropic/claude-sonnet-5",
            "cohere/command-a-03-2025",
            "deepseek/deepseek-v4-flash-0731",
            "deepseek/deepseek-v4-pro",
            "deepseek/deepseek-v4-pro-0813",
            "fireworks/deepseek-r1",
            "fireworks/deepseek-v3",
            "fireworks/deepseek-v3-0324",
            "fireworks/deepseek-v3p2-thinking",
            "fireworks/gpt-oss-120b",
            "fireworks/gpt-oss-20b",
            "fireworks/llama4-maverick-instruct-basic",
            "fireworks/qwen3-235b-a22b",
            "google/gemini-1.5-flash-002",
            "google/gemini-1.5-pro-002",
            "google/gemini-2.0-flash-001",
            "google/gemini-2.0-flash-exp",
            "google/gemini-2.0-flash-thinking-exp-01-21",
            "google/gemini-2.0-pro-exp-02-05",
            "google/gemini-2.5-flash-lite-preview-09-2025",
            "google/gemini-2.5-flash-lite-preview-09-2025-thinking",
            "google/gemini-2.5-flash-preview-04-17",
            "google/gemini-2.5-flash-preview-04-17-thinking",
            "google/gemini-2.5-flash-preview-09-2025",
            "google/gemini-2.5-flash-preview-09-2025-thinking",
            "google/gemini-2.5-pro-exp-03-25",
            "google/gemini-3-flash-preview",
            "google/gemini-3-pro-preview",
            "google/gemini-3.1-flash-lite-preview",
            "google/gemini-3.1-pro-preview",
            "google/gemini-3.5-flash",
            "google/gemini-3.5-flash-lite",
            "google/gemini-3.6-flash",
            "google/gemini-3.7-flash",
            "google/gemini-3.8-flash",
            "grok/grok-2-1212",
            "grok/grok-3",
            "grok/grok-3-mini-fast-high-reasoning",
            "grok/grok-3-mini-fast-low-reasoning",
            "grok/grok-4-0709",
            "grok/grok-4-1-fast-non-reasoning",
            "grok/grok-4-1-fast-reasoning",
            "grok/grok-4-fast-non-reasoning",
            "grok/grok-4-fast-reasoning",
            "grok/grok-4.20-0309-reasoning",
            "grok/grok-4.3",
            "grok/grok-4.5",
            "grok/grok-4.6",
            "kimi/kimi-k2-thinking",
            "kimi/kimi-k2.5-thinking",
            "kimi/kimi-k2.6",
            "kimi/kimi-k3",
            "meta/muse_spark",
            "meta/muse_spark_1_1",
            "meta/muse_spark_1_2",
            "minimax/MiniMax-M2.1",
            "minimax/MiniMax-M2.5",
            "minimax/MiniMax-M2.7",
            "minimax/MiniMax-M3",
            "mistralai/magistral-medium-2509",
            "mistralai/magistral-small-2509",
            "mistralai/mistral-large-2411",
            "mistralai/mistral-large-2512",
            "mistralai/mistral-medium-2505",
            "mistralai/mistral-medium-3.5",
            "mistralai/mistral-small-2402",
            "mistralai/mistral-small-2503",
            "nvidia/nemotron-3-ultra-550b-a55b",
            "openai/gpt-4.1-2025-04-14",
            "openai/gpt-4.1-mini-2025-04-14",
            "openai/gpt-4.1-nano-2025-04-14",
            "openai/gpt-4o-2024-08-06",
            "openai/gpt-4o-2024-11-20",
            "openai/gpt-4o-mini-2024-07-18",
            "openai/gpt-5-2025-08-07",
            "openai/gpt-5-mini-2025-08-07",
            "openai/gpt-5-nano-2025-08-07",
            "openai/gpt-5.1-2025-11-13",
            "openai/gpt-5.2-2025-12-11",
            "openai/gpt-5.4-2026-03-05",
            "openai/gpt-5.4-mini-2026-03-17",
            "openai/gpt-5.4-nano-2026-03-17",
            "openai/gpt-5.5",
            "openai/gpt-5.6-luna",
            "openai/gpt-5.6-sol",
            "openai/gpt-5.6-terra",
            "openai/o1-2024-12-17",
            "openai/o3-2025-04-16",
            "openai/o3-mini-2025-01-31",
            "openai/o4-mini-2025-04-16",
            "poolside/laguna-m.1",
            "poolside/laguna-xs.2",
            "thinkingmachines/inkling",
            "thinkingmachines/inkling-small",
            "together/langston/nim/nvidia/llama-3.3-nemotron-super-49b-v1-42e84561",
            "together/langston/nim/nvidia/llama-3.3-nemotron-super-49b-v1-42e84561-thinking",
            "together/meta-llama/Llama-3.3-70B-Instruct-Turbo",
            "together/meta-llama/Llama-4-Scout-17B-16E-Instruct",
            "together/meta-llama/Meta-Llama-3.1-405B-Instruct-Turbo",
            "together/meta-llama/Meta-Llama-3.1-70B-Instruct-Turbo",
            "together/meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo",
            "together/moonshotai/Kimi-K2-Instruct",
            "xiaomi/mimo-v2.5",
            "xiaomi/mimo-v2.5-pro",
            "zai/glm-4.5",
            "zai/glm-4.6",
            "zai/glm-4.7",
            "zai/glm-5-thinking",
            "zai/glm-5.1",
            "zai/glm-5.2",
            "zai/glm-5.3",
            "zai/glm-5.3-flash"
        ],
        "partners": [],
        "showBadge": false,
        "visible": true,
        "use_cost_per_test": false,
        "runner": "platform",
        "mode": "one-shot",
        "archived": false,
        "partner": false,
        "total_models": 145
    },
    "tasks": {
        "overall": {
            "meta/muse_spark_1_2": {
                "accuracy": 80.376,
                "latency": 48.313,
                "stderr": 0.763,
                "cost_per_test": 0.014199,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "meta/muse_spark_1_1": {
                "accuracy": 79.722,
                "latency": 32.816,
                "stderr": 0.77,
                "cost_per_test": 0.014921,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "meta/muse_spark": {
                "accuracy": 77.678,
                "latency": 57.522,
                "stderr": 0.806,
                "cost_per_test": 0.001922,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 77.106,
                "latency": 127.879,
                "stderr": 0.818,
                "cost_per_test": 0.127973,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-fable-5": {
                "accuracy": 76.942,
                "latency": 56.865,
                "stderr": 0.821,
                "cost_per_test": 0.220133,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.6-terra": {
                "accuracy": 76.166,
                "latency": 37.28,
                "stderr": 0.836,
                "cost_per_test": 0.035909,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.6-luna": {
                "accuracy": 76.166,
                "latency": 76.52,
                "stderr": 0.839,
                "cost_per_test": 0.012217,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 75.961,
                "latency": 79.254,
                "stderr": 0.834,
                "cost_per_test": 0.071368,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-fable-5-1": {
                "accuracy": 75.961,
                "latency": 85.564,
                "stderr": 0.83,
                "cost_per_test": 0.346549,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-3": {
                "accuracy": 75.879,
                "latency": 13.058,
                "stderr": 0.83,
                "cost_per_test": 0.017475,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 75.756,
                "latency": 60.479,
                "stderr": 0.845,
                "cost_per_test": 0.066435,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "kimi/kimi-k3": {
                "accuracy": 75.716,
                "latency": 144.708,
                "stderr": 0.841,
                "cost_per_test": 0.078156,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "grok/grok-4-fast-reasoning": {
                "accuracy": 75.697,
                "latency": 10.184,
                "stderr": 0.76,
                "cost_per_test": 0.001268,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-opus-4-8": {
                "accuracy": 75.634,
                "latency": 70.33,
                "stderr": 0.838,
                "cost_per_test": 0.159139,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-sonnet-5": {
                "accuracy": 75.634,
                "latency": 202.488,
                "stderr": 0.836,
                "cost_per_test": 0.301619,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "zai/glm-5.3-flash": {
                "accuracy": 75.593,
                "latency": 53.06,
                "stderr": 0.836,
                "cost_per_test": 0.001505,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "alibaba/qwen3.8-max": {
                "accuracy": 75.552,
                "latency": 265.751,
                "stderr": 0.838,
                "cost_per_test": 0.07645,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "thinkingmachines/inkling-small": {
                "accuracy": 75.511,
                "latency": 241.592,
                "stderr": 0.837,
                "cost_per_test": 0.017501,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 262000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "alibaba/qwen3.7-max": {
                "accuracy": 75.306,
                "latency": 57.677,
                "stderr": 0.839,
                "cost_per_test": 0.028122,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "thinkingmachines/inkling": {
                "accuracy": 75.306,
                "latency": 197.78,
                "stderr": 0.839,
                "cost_per_test": 0.103224,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "anthropic/claude-opus-4-7": {
                "accuracy": 75.266,
                "latency": 62.071,
                "stderr": 0.841,
                "cost_per_test": 0.124592,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 75.225,
                "latency": 37.434,
                "stderr": 0.847,
                "cost_per_test": 0.011528,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-5": {
                "accuracy": 75.143,
                "latency": 58.339,
                "stderr": 0.832,
                "cost_per_test": 0.114274,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-4.1-2025-04-14": {
                "accuracy": 75.061,
                "latency": 7.89,
                "stderr": 0.845,
                "cost_per_test": 0.006214,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 74.98,
                "latency": 95.83,
                "stderr": 0.854,
                "cost_per_test": 0.06042,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3.6-flash": {
                "accuracy": 74.857,
                "latency": 12.335,
                "stderr": 0.847,
                "cost_per_test": 0.020385,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 74.857,
                "latency": 44.176,
                "stderr": 0.854,
                "cost_per_test": 0.007193,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 74.856,
                "latency": 47.448,
                "stderr": 0.849,
                "cost_per_test": 0.218987,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/o4-mini-2025-04-16": {
                "accuracy": 74.776,
                "latency": 15.447,
                "stderr": 0.853,
                "cost_per_test": 0.007057,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.6-sol": {
                "accuracy": 74.775,
                "latency": 62.034,
                "stderr": 0.856,
                "cost_per_test": 0.138418,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3.7-flash": {
                "accuracy": 74.734,
                "latency": 7.832,
                "stderr": 0.85,
                "cost_per_test": 0.018681,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 74.734,
                "latency": 70.634,
                "stderr": 0.852,
                "cost_per_test": 0.01145,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 74.652,
                "latency": 267.484,
                "stderr": 0.846,
                "cost_per_test": 0.029274,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "openai/o3-2025-04-16": {
                "accuracy": 74.571,
                "latency": 24.024,
                "stderr": 0.854,
                "cost_per_test": 0.012997,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-4o-2024-11-20": {
                "accuracy": 74.53,
                "latency": 5.855,
                "stderr": 0.859,
                "cost_per_test": 0.005806,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3.8-flash": {
                "accuracy": 74.448,
                "latency": 8.759,
                "stderr": 0.848,
                "cost_per_test": 0.023535,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.5-flash": {
                "accuracy": 74.366,
                "latency": 19.694,
                "stderr": 0.846,
                "cost_per_test": 0.038136,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101": {
                "accuracy": 74.325,
                "latency": 14.876,
                "stderr": 0.854,
                "cost_per_test": 0.058665,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/o1-2024-12-17": {
                "accuracy": 74.284,
                "latency": 19.477,
                "stderr": 0.861,
                "cost_per_test": 0.12173,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 74.202,
                "latency": 90.464,
                "stderr": 0.846,
                "cost_per_test": 0.014221,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 74.121,
                "latency": 13.394,
                "stderr": 0.859,
                "cost_per_test": 0.017152,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-3-7-sonnet-20250219-thinking": {
                "accuracy": 74.039,
                "latency": 43.314,
                "stderr": 0.859,
                "cost_per_test": 0.050764,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 18384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3-max-preview": {
                "accuracy": 73.958,
                "latency": 21.405,
                "stderr": 0.856,
                "cost_per_test": 0.00671,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 73.958,
                "latency": 78.717,
                "stderr": 0.868,
                "cost_per_test": 0.157344,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 73.876,
                "latency": 12.001,
                "stderr": 0.857,
                "cost_per_test": 0.005764,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "xiaomi/mimo-v2.5-pro": {
                "accuracy": 73.794,
                "latency": 62.829,
                "stderr": 0.863,
                "cost_per_test": 0.002547,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "anthropic/claude-opus-4-1-20250805-thinking": {
                "accuracy": 73.672,
                "latency": 26.637,
                "stderr": 0.864,
                "cost_per_test": 0.093482,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3-max": {
                "accuracy": 73.508,
                "latency": 29.296,
                "stderr": 0.86,
                "cost_per_test": 0.00654,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 73.385,
                "latency": 78.668,
                "stderr": 0.868,
                "cost_per_test": 0.053873,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-5.2": {
                "accuracy": 73.344,
                "latency": 64.942,
                "stderr": 0.865,
                "cost_per_test": 0.015431,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 73.303,
                "latency": 48.391,
                "stderr": 0.861,
                "cost_per_test": 0.040814,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 73.14,
                "latency": 30.758,
                "stderr": 0.865,
                "cost_per_test": 0.001333,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "nvidia/nemotron-3-ultra-550b-a55b": {
                "accuracy": 73.099,
                "latency": 36.485,
                "stderr": 0.866,
                "cost_per_test": null,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Nvidia",
                "harness": null
            },
            "mistralai/mistral-large-2512": {
                "accuracy": 73.058,
                "latency": 29.531,
                "stderr": 0.867,
                "cost_per_test": 0.001601,
                "temperature": 0.065,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "deepseek/deepseek-v4-pro-0813": {
                "accuracy": 73.058,
                "latency": 226.296,
                "stderr": 0.872,
                "cost_per_test": 0.045816,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "grok/grok-3-mini-fast-low-reasoning": {
                "accuracy": 72.976,
                "latency": 11.728,
                "stderr": 0.87,
                "cost_per_test": 0.00112,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "low",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-2.5-pro-exp-03-25": {
                "accuracy": 72.894,
                "latency": 21.775,
                "stderr": 0.869,
                "cost_per_test": 0.010628,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 72.882,
                "latency": 41.931,
                "stderr": 0.863,
                "cost_per_test": 0.03149,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025": {
                "accuracy": 72.731,
                "latency": 6.267,
                "stderr": 0.869,
                "cost_per_test": 0.003098,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "minimax/MiniMax-M3": {
                "accuracy": 72.731,
                "latency": 94.036,
                "stderr": 0.863,
                "cost_per_test": 0.00728,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 512000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "google/gemini-3.5-flash-lite": {
                "accuracy": 72.608,
                "latency": 8.815,
                "stderr": 0.875,
                "cost_per_test": 0.00729,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 72.568,
                "latency": 24.404,
                "stderr": 0.872,
                "cost_per_test": 0.034377,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-3-7-sonnet-20250219": {
                "accuracy": 72.404,
                "latency": 6.28,
                "stderr": 0.879,
                "cost_per_test": 0.006427,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025-thinking": {
                "accuracy": 72.404,
                "latency": 10.635,
                "stderr": 0.872,
                "cost_per_test": 0.003099,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "zai/glm-4.5": {
                "accuracy": 72.404,
                "latency": 129.388,
                "stderr": 0.872,
                "cost_per_test": 0.009559,
                "temperature": 0.6,
                "top_p": 1.0,
                "max_output_tokens": 81920,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "zai/glm-5.3": {
                "accuracy": 72.363,
                "latency": 98.323,
                "stderr": 0.879,
                "cost_per_test": 0.032,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "fireworks/deepseek-r1": {
                "accuracy": 72.281,
                "latency": 152.958,
                "stderr": 0.874,
                "cost_per_test": 0.012239,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "alibaba/qwen3.5-flash": {
                "accuracy": 72.158,
                "latency": 73.003,
                "stderr": 0.876,
                "cost_per_test": 0.003099,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 72.077,
                "latency": 189.695,
                "stderr": 0.877,
                "cost_per_test": 0.033187,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514-thinking": {
                "accuracy": 71.995,
                "latency": 32.385,
                "stderr": 0.878,
                "cost_per_test": 0.029421,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-4.1-mini-2025-04-14": {
                "accuracy": 71.914,
                "latency": 5.921,
                "stderr": 0.881,
                "cost_per_test": 0.001233,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-4-20250514": {
                "accuracy": 71.914,
                "latency": 15.23,
                "stderr": 0.879,
                "cost_per_test": 0.03694,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "xiaomi/mimo-v2.5": {
                "accuracy": 71.832,
                "latency": 26.3,
                "stderr": 0.881,
                "cost_per_test": 0.000729,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 71.79,
                "latency": 7.644,
                "stderr": 0.88,
                "cost_per_test": 0.00096,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "kimi/kimi-k2-thinking": {
                "accuracy": 71.709,
                "latency": 61.408,
                "stderr": 0.878,
                "cost_per_test": 0.00913,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "grok/grok-4.5": {
                "accuracy": 71.668,
                "latency": 195.458,
                "stderr": 0.892,
                "cost_per_test": 0.019211,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "fireworks/gpt-oss-120b": {
                "accuracy": 71.586,
                "latency": 47.661,
                "stderr": 0.878,
                "cost_per_test": 0.002958,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "grok/grok-4-fast-non-reasoning": {
                "accuracy": 71.581,
                "latency": 4.25,
                "stderr": 0.783,
                "cost_per_test": 0.000571,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-opus-4-1-20250805": {
                "accuracy": 71.464,
                "latency": 32.602,
                "stderr": 0.882,
                "cost_per_test": 0.039842,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3.6-27b": {
                "accuracy": 71.26,
                "latency": 147.778,
                "stderr": 0.882,
                "cost_per_test": 0.030256,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 71.218,
                "latency": 44.578,
                "stderr": 0.9,
                "cost_per_test": 0.00677,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-5.1": {
                "accuracy": 71.194,
                "latency": 84.072,
                "stderr": 0.895,
                "cost_per_test": 0.016958,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-04-17": {
                "accuracy": 71.178,
                "latency": 5.388,
                "stderr": 0.885,
                "cost_per_test": 0.00261,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-4o-2024-08-06": {
                "accuracy": 71.136,
                "latency": 9.543,
                "stderr": 0.892,
                "cost_per_test": 0.004417,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-3-mini-fast-high-reasoning": {
                "accuracy": 71.136,
                "latency": 25.447,
                "stderr": 0.889,
                "cost_per_test": 0.002142,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "fireworks/deepseek-v3-0324": {
                "accuracy": 71.096,
                "latency": 34.009,
                "stderr": 0.883,
                "cost_per_test": 0.000652,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "grok/grok-4.6": {
                "accuracy": 71.096,
                "latency": 47.57,
                "stderr": 0.899,
                "cost_per_test": 0.017298,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "alibaba/qwen3.8-27b": {
                "accuracy": 70.85,
                "latency": 94.491,
                "stderr": 0.893,
                "cost_per_test": 0.046465,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "grok/grok-4.3": {
                "accuracy": 70.81,
                "latency": 391.972,
                "stderr": 0.896,
                "cost_per_test": 0.013656,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "deepseek/deepseek-v4-flash-0731": {
                "accuracy": 70.687,
                "latency": 72.203,
                "stderr": 0.897,
                "cost_per_test": 0.009755,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "ant/ling-3.0-flash-2607": {
                "accuracy": 70.646,
                "latency": 7.54,
                "stderr": 0.895,
                "cost_per_test": 0.000664,
                "temperature": 0.6,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "fireworks/qwen3-235b-a22b": {
                "accuracy": 70.646,
                "latency": 92.742,
                "stderr": 0.89,
                "cost_per_test": 0.004644,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-04-17-thinking": {
                "accuracy": 70.524,
                "latency": 15.355,
                "stderr": 0.891,
                "cost_per_test": 0.007461,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "mistralai/mistral-medium-2505": {
                "accuracy": 70.319,
                "latency": 11.052,
                "stderr": 0.894,
                "cost_per_test": 0.001422,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "together/moonshotai/Kimi-K2-Instruct": {
                "accuracy": 70.196,
                "latency": 12.834,
                "stderr": 0.895,
                "cost_per_test": 0.001543,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "anthropic/claude-3-5-sonnet-20241022": {
                "accuracy": 70.156,
                "latency": 5.943,
                "stderr": 0.902,
                "cost_per_test": 0.006006,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "zai/glm-5-thinking": {
                "accuracy": 70.033,
                "latency": 149.298,
                "stderr": 0.9,
                "cost_per_test": 0.021913,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "google/gemini-2.0-flash-thinking-exp-01-21": {
                "accuracy": 69.788,
                "latency": 12.807,
                "stderr": 0.899,
                "cost_per_test": 0.000827,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514": {
                "accuracy": 69.624,
                "latency": 11.847,
                "stderr": 0.899,
                "cost_per_test": 0.0076,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/o3-mini-2025-01-31": {
                "accuracy": 69.42,
                "latency": 74.312,
                "stderr": 0.909,
                "cost_per_test": 0.018142,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-4.7": {
                "accuracy": 68.766,
                "latency": 187.181,
                "stderr": 0.921,
                "cost_per_test": 0.013873,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "google/gemini-2.0-pro-exp-02-05": {
                "accuracy": 68.152,
                "latency": 8.989,
                "stderr": 0.916,
                "cost_per_test": 0.004012,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "minimax/MiniMax-M2.5": {
                "accuracy": 68.152,
                "latency": 73.243,
                "stderr": 0.906,
                "cost_per_test": 0.008629,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "fireworks/deepseek-v3p2-thinking": {
                "accuracy": 68.152,
                "latency": 98.919,
                "stderr": 0.919,
                "cost_per_test": 0.002376,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 20480,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "mistralai/mistral-medium-3.5": {
                "accuracy": 67.988,
                "latency": 102.392,
                "stderr": 0.916,
                "cost_per_test": 0.079101,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "fireworks/deepseek-v3": {
                "accuracy": 67.907,
                "latency": 27.696,
                "stderr": 0.918,
                "cost_per_test": 0.00056,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "google/gemini-2.0-flash-exp": {
                "accuracy": 67.744,
                "latency": 7.486,
                "stderr": 0.915,
                "cost_per_test": 0.000302,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 67.539,
                "latency": 47.524,
                "stderr": 0.914,
                "cost_per_test": 0.02743,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 67.416,
                "latency": 15.898,
                "stderr": 0.926,
                "cost_per_test": 0.000638,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5-nano-2025-08-07": {
                "accuracy": 67.376,
                "latency": 69.944,
                "stderr": 0.913,
                "cost_per_test": 0.004904,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-2-1212": {
                "accuracy": 67.048,
                "latency": 10.621,
                "stderr": 0.918,
                "cost_per_test": 0.006894,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 0,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "fireworks/llama4-maverick-instruct-basic": {
                "accuracy": 66.558,
                "latency": 25.143,
                "stderr": 0.919,
                "cost_per_test": 0.000527,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 66.558,
                "latency": 28.694,
                "stderr": 0.92,
                "cost_per_test": 0.003474,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "minimax/MiniMax-M2.1": {
                "accuracy": 66.353,
                "latency": 113.258,
                "stderr": 0.925,
                "cost_per_test": 0.009137,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "zai/glm-4.6": {
                "accuracy": 66.235,
                "latency": 64.149,
                "stderr": 0.808,
                "cost_per_test": 0.00928,
                "temperature": 0.6,
                "top_p": 1.0,
                "max_output_tokens": 122880,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025": {
                "accuracy": 66.231,
                "latency": 3.597,
                "stderr": 0.923,
                "cost_per_test": 0.000477,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-2.0-flash-001": {
                "accuracy": 65.25,
                "latency": 5.702,
                "stderr": 0.933,
                "cost_per_test": 0.000296,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-4-0709": {
                "accuracy": 65.086,
                "latency": 159.421,
                "stderr": 0.962,
                "cost_per_test": 0.077989,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025-thinking": {
                "accuracy": 64.718,
                "latency": 3.94,
                "stderr": 0.931,
                "cost_per_test": 0.000509,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "mistralai/mistral-large-2411": {
                "accuracy": 63.778,
                "latency": 12.019,
                "stderr": 0.95,
                "cost_per_test": 0.004079,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "together/langston/nim/nvidia/llama-3.3-nemotron-super-49b-v1-42e84561-thinking": {
                "accuracy": 63.736,
                "latency": 32.67,
                "stderr": 0.943,
                "cost_per_test": null,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "fireworks/gpt-oss-20b": {
                "accuracy": 63.696,
                "latency": 49.004,
                "stderr": 0.936,
                "cost_per_test": 0.002254,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "mistralai/magistral-medium-2509": {
                "accuracy": 61.938,
                "latency": 46.254,
                "stderr": 0.964,
                "cost_per_test": 0.025214,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "grok/grok-4-1-fast-non-reasoning": {
                "accuracy": 61.856,
                "latency": 9.277,
                "stderr": 0.914,
                "cost_per_test": 0.000372,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "cohere/command-a-03-2025": {
                "accuracy": 61.366,
                "latency": 11.214,
                "stderr": 0.952,
                "cost_per_test": 0.006484,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 8000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "ai21labs/jamba-large-1.6": {
                "accuracy": 60.875,
                "latency": 16.439,
                "stderr": 0.949,
                "cost_per_test": 0.005162,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "together/meta-llama/Meta-Llama-3.1-405B-Instruct-Turbo": {
                "accuracy": 60.875,
                "latency": 23.551,
                "stderr": 0.959,
                "cost_per_test": 0.001904,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "openai/gpt-4.1-nano-2025-04-14": {
                "accuracy": 60.752,
                "latency": 2.998,
                "stderr": 0.956,
                "cost_per_test": 0.000291,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-4o-mini-2024-07-18": {
                "accuracy": 60.548,
                "latency": 8.899,
                "stderr": 0.963,
                "cost_per_test": 0.000323,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "mistralai/magistral-small-2509": {
                "accuracy": 60.302,
                "latency": 16.441,
                "stderr": 0.967,
                "cost_per_test": 0.003909,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "together/langston/nim/nvidia/llama-3.3-nemotron-super-49b-v1-42e84561": {
                "accuracy": 60.22,
                "latency": 18.308,
                "stderr": 0.959,
                "cost_per_test": null,
                "temperature": 0.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "google/gemini-1.5-pro-002": {
                "accuracy": 59.485,
                "latency": 9.86,
                "stderr": 0.962,
                "cost_per_test": 0.002545,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "together/meta-llama/Llama-3.3-70B-Instruct-Turbo": {
                "accuracy": 59.444,
                "latency": 3.836,
                "stderr": 0.962,
                "cost_per_test": 0.000532,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "poolside/laguna-xs.2": {
                "accuracy": 58.953,
                "latency": 27.449,
                "stderr": 0.964,
                "cost_per_test": 0.000305,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "mistralai/mistral-small-2503": {
                "accuracy": 58.3,
                "latency": 7.952,
                "stderr": 0.968,
                "cost_per_test": 0.000184,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "ai21labs/jamba-1.5-large": {
                "accuracy": 58.176,
                "latency": 20.028,
                "stderr": 0.971,
                "cost_per_test": 0.005384,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "anthropic/claude-3-5-haiku-20241022": {
                "accuracy": 57.359,
                "latency": 4.987,
                "stderr": 0.976,
                "cost_per_test": 0.001317,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "together/meta-llama/Meta-Llama-3.1-70B-Instruct-Turbo": {
                "accuracy": 56.174,
                "latency": 4.327,
                "stderr": 0.973,
                "cost_per_test": 0.000488,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "together/meta-llama/Llama-4-Scout-17B-16E-Instruct": {
                "accuracy": 55.192,
                "latency": 7.042,
                "stderr": 0.975,
                "cost_per_test": 0.000378,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "mistralai/mistral-small-2402": {
                "accuracy": 49.142,
                "latency": 9.018,
                "stderr": 0.987,
                "cost_per_test": 0.000355,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "google/gemini-1.5-flash-002": {
                "accuracy": 48.202,
                "latency": 3.701,
                "stderr": 0.98,
                "cost_per_test": 0.000155,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "ai21labs/jamba-mini-1.6": {
                "accuracy": 44.604,
                "latency": 4.337,
                "stderr": 0.963,
                "cost_per_test": 0.00026,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "ai21labs/jamba-1.5-mini": {
                "accuracy": 41.864,
                "latency": 4.78,
                "stderr": 0.967,
                "cost_per_test": 0.00028,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "together/meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo": {
                "accuracy": 32.338,
                "latency": 2.458,
                "stderr": 0.92,
                "cost_per_test": 0.000107,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "poolside/laguna-m.1": {
                "accuracy": 1.636,
                "latency": 212.643,
                "stderr": 0.256,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            }
        },
        "correctness": {
            "grok/grok-4-fast-reasoning": {
                "accuracy": 68.244,
                "latency": 10.184,
                "stderr": 1.185,
                "cost_per_test": 0.001268,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "meta/muse_spark_1_2": {
                "accuracy": 68.029,
                "latency": 48.313,
                "stderr": 1.334,
                "cost_per_test": 0.014199,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "meta/muse_spark_1_1": {
                "accuracy": 66.885,
                "latency": 32.816,
                "stderr": 1.346,
                "cost_per_test": 0.014921,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "openai/gpt-5.6-luna": {
                "accuracy": 66.558,
                "latency": 76.52,
                "stderr": 1.349,
                "cost_per_test": 0.012217,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 66.312,
                "latency": 60.479,
                "stderr": 1.352,
                "cost_per_test": 0.066435,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.6-terra": {
                "accuracy": 65.822,
                "latency": 37.28,
                "stderr": 1.356,
                "cost_per_test": 0.035909,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-fable-5": {
                "accuracy": 65.822,
                "latency": 56.865,
                "stderr": 1.356,
                "cost_per_test": 0.220133,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 65.822,
                "latency": 127.879,
                "stderr": 1.356,
                "cost_per_test": 0.127973,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "meta/muse_spark": {
                "accuracy": 65.576,
                "latency": 57.522,
                "stderr": 1.359,
                "cost_per_test": 0.001922,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 65.495,
                "latency": 95.83,
                "stderr": 1.359,
                "cost_per_test": 0.06042,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "kimi/kimi-k3": {
                "accuracy": 65.331,
                "latency": 144.708,
                "stderr": 1.361,
                "cost_per_test": 0.078156,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "openai/gpt-5.6-sol": {
                "accuracy": 65.004,
                "latency": 62.034,
                "stderr": 1.364,
                "cost_per_test": 0.138418,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 64.922,
                "latency": 78.717,
                "stderr": 1.365,
                "cost_per_test": 0.157344,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-4o-2024-11-20": {
                "accuracy": 64.841,
                "latency": 5.855,
                "stderr": 1.365,
                "cost_per_test": 0.005806,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 64.841,
                "latency": 44.176,
                "stderr": 1.365,
                "cost_per_test": 0.007193,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 64.759,
                "latency": 37.434,
                "stderr": 1.366,
                "cost_per_test": 0.011528,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 64.759,
                "latency": 79.254,
                "stderr": 1.366,
                "cost_per_test": 0.071368,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-4-8": {
                "accuracy": 64.513,
                "latency": 70.33,
                "stderr": 1.368,
                "cost_per_test": 0.159139,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/o4-mini-2025-04-16": {
                "accuracy": 64.432,
                "latency": 15.447,
                "stderr": 1.369,
                "cost_per_test": 0.007057,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/o1-2024-12-17": {
                "accuracy": 64.432,
                "latency": 19.477,
                "stderr": 1.369,
                "cost_per_test": 0.12173,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 64.186,
                "latency": 70.634,
                "stderr": 1.371,
                "cost_per_test": 0.01145,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "anthropic/claude-fable-5-1": {
                "accuracy": 64.105,
                "latency": 85.564,
                "stderr": 1.372,
                "cost_per_test": 0.346549,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3.8-max": {
                "accuracy": 64.105,
                "latency": 265.751,
                "stderr": 1.372,
                "cost_per_test": 0.07645,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/o3-2025-04-16": {
                "accuracy": 64.023,
                "latency": 24.024,
                "stderr": 1.372,
                "cost_per_test": 0.012997,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-5.3-flash": {
                "accuracy": 64.023,
                "latency": 53.06,
                "stderr": 1.372,
                "cost_per_test": 0.001505,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "anthropic/claude-sonnet-5": {
                "accuracy": 64.023,
                "latency": 202.488,
                "stderr": 1.372,
                "cost_per_test": 0.301619,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-4.1-2025-04-14": {
                "accuracy": 63.859,
                "latency": 7.89,
                "stderr": 1.374,
                "cost_per_test": 0.006214,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-3": {
                "accuracy": 63.859,
                "latency": 13.058,
                "stderr": 1.374,
                "cost_per_test": 0.017475,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 63.859,
                "latency": 47.448,
                "stderr": 1.374,
                "cost_per_test": 0.218987,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "thinkingmachines/inkling-small": {
                "accuracy": 63.859,
                "latency": 241.592,
                "stderr": 1.374,
                "cost_per_test": 0.017501,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 262000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "anthropic/claude-opus-4-7": {
                "accuracy": 63.778,
                "latency": 62.071,
                "stderr": 1.374,
                "cost_per_test": 0.124592,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.7-flash": {
                "accuracy": 63.614,
                "latency": 7.832,
                "stderr": 1.376,
                "cost_per_test": 0.018681,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.6-flash": {
                "accuracy": 63.532,
                "latency": 12.335,
                "stderr": 1.376,
                "cost_per_test": 0.020385,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3.7-max": {
                "accuracy": 63.532,
                "latency": 57.677,
                "stderr": 1.376,
                "cost_per_test": 0.028122,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "thinkingmachines/inkling": {
                "accuracy": 63.532,
                "latency": 197.78,
                "stderr": 1.376,
                "cost_per_test": 0.103224,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 63.369,
                "latency": 13.394,
                "stderr": 1.378,
                "cost_per_test": 0.017152,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-3-7-sonnet-20250219-thinking": {
                "accuracy": 63.287,
                "latency": 43.314,
                "stderr": 1.378,
                "cost_per_test": 0.050764,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 18384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101": {
                "accuracy": 63.123,
                "latency": 14.876,
                "stderr": 1.38,
                "cost_per_test": 0.058665,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "xiaomi/mimo-v2.5-pro": {
                "accuracy": 63.123,
                "latency": 62.829,
                "stderr": 1.38,
                "cost_per_test": 0.002547,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 63.042,
                "latency": 44.578,
                "stderr": 1.38,
                "cost_per_test": 0.00677,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-4-1-20250805-thinking": {
                "accuracy": 62.96,
                "latency": 26.637,
                "stderr": 1.381,
                "cost_per_test": 0.093482,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 62.796,
                "latency": 78.668,
                "stderr": 1.382,
                "cost_per_test": 0.053873,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 62.715,
                "latency": 267.484,
                "stderr": 1.383,
                "cost_per_test": 0.029274,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "deepseek/deepseek-v4-pro-0813": {
                "accuracy": 62.633,
                "latency": 226.296,
                "stderr": 1.383,
                "cost_per_test": 0.045816,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "grok/grok-4.5": {
                "accuracy": 62.551,
                "latency": 195.458,
                "stderr": 1.384,
                "cost_per_test": 0.019211,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-3.8-flash": {
                "accuracy": 62.469,
                "latency": 8.759,
                "stderr": 1.385,
                "cost_per_test": 0.023535,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3-max-preview": {
                "accuracy": 62.388,
                "latency": 21.405,
                "stderr": 1.385,
                "cost_per_test": 0.00671,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 62.306,
                "latency": 12.001,
                "stderr": 1.386,
                "cost_per_test": 0.005764,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-4.6": {
                "accuracy": 62.142,
                "latency": 47.57,
                "stderr": 1.387,
                "cost_per_test": 0.017298,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "zai/glm-5.2": {
                "accuracy": 62.061,
                "latency": 64.942,
                "stderr": 1.388,
                "cost_per_test": 0.015431,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "grok/grok-4-0709": {
                "accuracy": 62.061,
                "latency": 159.421,
                "stderr": 1.388,
                "cost_per_test": 0.077989,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "grok/grok-3-mini-fast-low-reasoning": {
                "accuracy": 61.979,
                "latency": 11.728,
                "stderr": 1.388,
                "cost_per_test": 0.00112,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "low",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-opus-5": {
                "accuracy": 61.979,
                "latency": 58.339,
                "stderr": 1.388,
                "cost_per_test": 0.114274,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "zai/glm-5.3": {
                "accuracy": 61.979,
                "latency": 98.323,
                "stderr": 1.388,
                "cost_per_test": 0.032,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "anthropic/claude-3-7-sonnet-20250219": {
                "accuracy": 61.897,
                "latency": 6.28,
                "stderr": 1.389,
                "cost_per_test": 0.006427,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.5-flash": {
                "accuracy": 61.897,
                "latency": 19.694,
                "stderr": 1.389,
                "cost_per_test": 0.038136,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-3.5-flash-lite": {
                "accuracy": 61.733,
                "latency": 8.815,
                "stderr": 1.39,
                "cost_per_test": 0.00729,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3-max": {
                "accuracy": 61.733,
                "latency": 29.296,
                "stderr": 1.39,
                "cost_per_test": 0.00654,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 61.652,
                "latency": 30.758,
                "stderr": 1.39,
                "cost_per_test": 0.001333,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "nvidia/nemotron-3-ultra-550b-a55b": {
                "accuracy": 61.652,
                "latency": 36.485,
                "stderr": 1.39,
                "cost_per_test": null,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Nvidia",
                "harness": null
            },
            "google/gemini-2.5-pro-exp-03-25": {
                "accuracy": 61.57,
                "latency": 21.775,
                "stderr": 1.391,
                "cost_per_test": 0.010628,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "mistralai/mistral-large-2512": {
                "accuracy": 61.57,
                "latency": 29.531,
                "stderr": 1.391,
                "cost_per_test": 0.001601,
                "temperature": 0.065,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "zai/glm-5.1": {
                "accuracy": 61.538,
                "latency": 84.072,
                "stderr": 1.392,
                "cost_per_test": 0.016958,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 61.488,
                "latency": 90.464,
                "stderr": 1.391,
                "cost_per_test": 0.014221,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 61.406,
                "latency": 48.391,
                "stderr": 1.392,
                "cost_per_test": 0.040814,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 61.079,
                "latency": 24.404,
                "stderr": 1.394,
                "cost_per_test": 0.034377,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025": {
                "accuracy": 60.998,
                "latency": 6.267,
                "stderr": 1.395,
                "cost_per_test": 0.003098,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-4.1-mini-2025-04-14": {
                "accuracy": 60.834,
                "latency": 5.921,
                "stderr": 1.396,
                "cost_per_test": 0.001233,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-4o-2024-08-06": {
                "accuracy": 60.834,
                "latency": 9.543,
                "stderr": 1.396,
                "cost_per_test": 0.004417,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "xiaomi/mimo-v2.5": {
                "accuracy": 60.752,
                "latency": 26.3,
                "stderr": 1.396,
                "cost_per_test": 0.000729,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "alibaba/qwen3.5-flash": {
                "accuracy": 60.752,
                "latency": 73.003,
                "stderr": 1.396,
                "cost_per_test": 0.003099,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "zai/glm-4.5": {
                "accuracy": 60.752,
                "latency": 129.388,
                "stderr": 1.396,
                "cost_per_test": 0.009559,
                "temperature": 0.6,
                "top_p": 1.0,
                "max_output_tokens": 81920,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "fireworks/deepseek-r1": {
                "accuracy": 60.752,
                "latency": 152.958,
                "stderr": 1.396,
                "cost_per_test": 0.012239,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 60.749,
                "latency": 41.931,
                "stderr": 1.393,
                "cost_per_test": 0.03149,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025-thinking": {
                "accuracy": 60.67,
                "latency": 10.635,
                "stderr": 1.397,
                "cost_per_test": 0.003099,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 60.589,
                "latency": 189.695,
                "stderr": 1.397,
                "cost_per_test": 0.033187,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "grok/grok-4.3": {
                "accuracy": 60.589,
                "latency": 391.972,
                "stderr": 1.397,
                "cost_per_test": 0.013656,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-opus-4-20250514": {
                "accuracy": 60.507,
                "latency": 15.23,
                "stderr": 1.398,
                "cost_per_test": 0.03694,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514-thinking": {
                "accuracy": 60.507,
                "latency": 32.385,
                "stderr": 1.398,
                "cost_per_test": 0.029421,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "deepseek/deepseek-v4-flash-0731": {
                "accuracy": 60.507,
                "latency": 72.203,
                "stderr": 1.398,
                "cost_per_test": 0.009755,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 60.425,
                "latency": 7.644,
                "stderr": 1.398,
                "cost_per_test": 0.00096,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "zai/glm-4.7": {
                "accuracy": 60.262,
                "latency": 187.181,
                "stderr": 1.399,
                "cost_per_test": 0.013873,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "grok/grok-3-mini-fast-high-reasoning": {
                "accuracy": 60.18,
                "latency": 25.447,
                "stderr": 1.4,
                "cost_per_test": 0.002142,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "alibaba/qwen3.8-27b": {
                "accuracy": 60.098,
                "latency": 94.491,
                "stderr": 1.4,
                "cost_per_test": 0.046465,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "minimax/MiniMax-M3": {
                "accuracy": 60.016,
                "latency": 94.036,
                "stderr": 1.401,
                "cost_per_test": 0.00728,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 512000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "ant/ling-3.0-flash-2607": {
                "accuracy": 59.935,
                "latency": 7.54,
                "stderr": 1.401,
                "cost_per_test": 0.000664,
                "temperature": 0.6,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "anthropic/claude-3-5-sonnet-20241022": {
                "accuracy": 59.853,
                "latency": 5.943,
                "stderr": 1.402,
                "cost_per_test": 0.006006,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-4-1-20250805": {
                "accuracy": 59.853,
                "latency": 32.602,
                "stderr": 1.402,
                "cost_per_test": 0.039842,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "kimi/kimi-k2-thinking": {
                "accuracy": 59.689,
                "latency": 61.408,
                "stderr": 1.403,
                "cost_per_test": 0.00913,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-04-17": {
                "accuracy": 59.608,
                "latency": 5.388,
                "stderr": 1.403,
                "cost_per_test": 0.00261,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-4-fast-non-reasoning": {
                "accuracy": 59.559,
                "latency": 4.25,
                "stderr": 1.249,
                "cost_per_test": 0.000571,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "fireworks/gpt-oss-120b": {
                "accuracy": 59.444,
                "latency": 47.661,
                "stderr": 1.404,
                "cost_per_test": 0.002958,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "openai/o3-mini-2025-01-31": {
                "accuracy": 59.362,
                "latency": 74.312,
                "stderr": 1.404,
                "cost_per_test": 0.018142,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3.6-27b": {
                "accuracy": 59.199,
                "latency": 147.778,
                "stderr": 1.405,
                "cost_per_test": 0.030256,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "zai/glm-5-thinking": {
                "accuracy": 59.199,
                "latency": 149.298,
                "stderr": 1.405,
                "cost_per_test": 0.021913,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "fireworks/qwen3-235b-a22b": {
                "accuracy": 59.035,
                "latency": 92.742,
                "stderr": 1.406,
                "cost_per_test": 0.004644,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "fireworks/deepseek-v3-0324": {
                "accuracy": 58.953,
                "latency": 34.009,
                "stderr": 1.407,
                "cost_per_test": 0.000652,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-04-17-thinking": {
                "accuracy": 58.872,
                "latency": 15.355,
                "stderr": 1.407,
                "cost_per_test": 0.007461,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "mistralai/mistral-medium-2505": {
                "accuracy": 58.79,
                "latency": 11.052,
                "stderr": 1.407,
                "cost_per_test": 0.001422,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "together/moonshotai/Kimi-K2-Instruct": {
                "accuracy": 58.79,
                "latency": 12.834,
                "stderr": 1.407,
                "cost_per_test": 0.001543,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "google/gemini-2.0-flash-thinking-exp-01-21": {
                "accuracy": 58.299,
                "latency": 12.807,
                "stderr": 1.41,
                "cost_per_test": 0.000827,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514": {
                "accuracy": 57.89,
                "latency": 11.847,
                "stderr": 1.412,
                "cost_per_test": 0.0076,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "fireworks/deepseek-v3p2-thinking": {
                "accuracy": 57.89,
                "latency": 98.919,
                "stderr": 1.412,
                "cost_per_test": 0.002376,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 20480,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 57.482,
                "latency": 15.898,
                "stderr": 1.414,
                "cost_per_test": 0.000638,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-2.0-pro-exp-02-05": {
                "accuracy": 57.318,
                "latency": 8.989,
                "stderr": 1.414,
                "cost_per_test": 0.004012,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "fireworks/deepseek-v3": {
                "accuracy": 57.073,
                "latency": 27.696,
                "stderr": 1.415,
                "cost_per_test": 0.00056,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "mistralai/mistral-medium-3.5": {
                "accuracy": 56.746,
                "latency": 102.392,
                "stderr": 1.417,
                "cost_per_test": 0.079101,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "google/gemini-2.0-flash-exp": {
                "accuracy": 56.092,
                "latency": 7.486,
                "stderr": 1.419,
                "cost_per_test": 0.000302,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "minimax/MiniMax-M2.5": {
                "accuracy": 55.437,
                "latency": 73.243,
                "stderr": 1.421,
                "cost_per_test": 0.008629,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 55.274,
                "latency": 47.524,
                "stderr": 1.422,
                "cost_per_test": 0.02743,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-2-1212": {
                "accuracy": 54.865,
                "latency": 10.621,
                "stderr": 1.423,
                "cost_per_test": 0.006894,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 0,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-5-nano-2025-08-07": {
                "accuracy": 54.702,
                "latency": 69.944,
                "stderr": 1.423,
                "cost_per_test": 0.004904,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "minimax/MiniMax-M2.1": {
                "accuracy": 54.456,
                "latency": 113.258,
                "stderr": 1.424,
                "cost_per_test": 0.009137,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 54.129,
                "latency": 28.694,
                "stderr": 1.425,
                "cost_per_test": 0.003474,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025": {
                "accuracy": 53.966,
                "latency": 3.597,
                "stderr": 1.425,
                "cost_per_test": 0.000477,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "fireworks/llama4-maverick-instruct-basic": {
                "accuracy": 53.966,
                "latency": 25.143,
                "stderr": 1.425,
                "cost_per_test": 0.000527,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "mistralai/mistral-large-2411": {
                "accuracy": 53.639,
                "latency": 12.019,
                "stderr": 1.426,
                "cost_per_test": 0.004079,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "google/gemini-2.0-flash-001": {
                "accuracy": 53.475,
                "latency": 5.702,
                "stderr": 1.426,
                "cost_per_test": 0.000296,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "mistralai/magistral-medium-2509": {
                "accuracy": 52.821,
                "latency": 46.254,
                "stderr": 1.427,
                "cost_per_test": 0.025214,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "together/langston/nim/nvidia/llama-3.3-nemotron-super-49b-v1-42e84561-thinking": {
                "accuracy": 52.085,
                "latency": 32.67,
                "stderr": 1.428,
                "cost_per_test": null,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025-thinking": {
                "accuracy": 52.003,
                "latency": 3.94,
                "stderr": 1.429,
                "cost_per_test": 0.000509,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "zai/glm-4.6": {
                "accuracy": 51.329,
                "latency": 64.149,
                "stderr": 1.272,
                "cost_per_test": 0.00928,
                "temperature": 0.6,
                "top_p": 1.0,
                "max_output_tokens": 122880,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "fireworks/gpt-oss-20b": {
                "accuracy": 50.695,
                "latency": 49.004,
                "stderr": 1.43,
                "cost_per_test": 0.002254,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "mistralai/magistral-small-2509": {
                "accuracy": 50.041,
                "latency": 16.441,
                "stderr": 1.43,
                "cost_per_test": 0.003909,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "openai/gpt-4o-mini-2024-07-18": {
                "accuracy": 49.55,
                "latency": 8.899,
                "stderr": 1.43,
                "cost_per_test": 0.000323,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "together/meta-llama/Meta-Llama-3.1-405B-Instruct-Turbo": {
                "accuracy": 49.305,
                "latency": 23.551,
                "stderr": 1.43,
                "cost_per_test": 0.001904,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "cohere/command-a-03-2025": {
                "accuracy": 48.978,
                "latency": 11.214,
                "stderr": 1.429,
                "cost_per_test": 0.006484,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 8000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "openai/gpt-4.1-nano-2025-04-14": {
                "accuracy": 48.569,
                "latency": 2.998,
                "stderr": 1.429,
                "cost_per_test": 0.000291,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "together/langston/nim/nvidia/llama-3.3-nemotron-super-49b-v1-42e84561": {
                "accuracy": 48.078,
                "latency": 18.308,
                "stderr": 1.429,
                "cost_per_test": null,
                "temperature": 0.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "ai21labs/jamba-large-1.6": {
                "accuracy": 47.506,
                "latency": 16.439,
                "stderr": 1.428,
                "cost_per_test": 0.005162,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "together/meta-llama/Llama-3.3-70B-Instruct-Turbo": {
                "accuracy": 47.343,
                "latency": 3.836,
                "stderr": 1.428,
                "cost_per_test": 0.000532,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "google/gemini-1.5-pro-002": {
                "accuracy": 47.343,
                "latency": 9.86,
                "stderr": 1.428,
                "cost_per_test": 0.002545,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "ai21labs/jamba-1.5-large": {
                "accuracy": 46.77,
                "latency": 20.028,
                "stderr": 1.427,
                "cost_per_test": 0.005384,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "poolside/laguna-xs.2": {
                "accuracy": 46.688,
                "latency": 27.449,
                "stderr": 1.427,
                "cost_per_test": 0.000305,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "anthropic/claude-3-5-haiku-20241022": {
                "accuracy": 46.525,
                "latency": 4.987,
                "stderr": 1.426,
                "cost_per_test": 0.001317,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "mistralai/mistral-small-2503": {
                "accuracy": 46.525,
                "latency": 7.952,
                "stderr": 1.426,
                "cost_per_test": 0.000184,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "together/meta-llama/Meta-Llama-3.1-70B-Instruct-Turbo": {
                "accuracy": 44.154,
                "latency": 4.327,
                "stderr": 1.42,
                "cost_per_test": 0.000488,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "grok/grok-4-1-fast-non-reasoning": {
                "accuracy": 44.072,
                "latency": 9.277,
                "stderr": 1.42,
                "cost_per_test": 0.000372,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "together/meta-llama/Llama-4-Scout-17B-16E-Instruct": {
                "accuracy": 43.009,
                "latency": 7.042,
                "stderr": 1.416,
                "cost_per_test": 0.000378,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "mistralai/mistral-small-2402": {
                "accuracy": 38.267,
                "latency": 9.018,
                "stderr": 1.39,
                "cost_per_test": 0.000355,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "google/gemini-1.5-flash-002": {
                "accuracy": 36.141,
                "latency": 3.701,
                "stderr": 1.374,
                "cost_per_test": 0.000155,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "ai21labs/jamba-mini-1.6": {
                "accuracy": 30.335,
                "latency": 4.337,
                "stderr": 1.315,
                "cost_per_test": 0.00026,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "ai21labs/jamba-1.5-mini": {
                "accuracy": 29.763,
                "latency": 4.78,
                "stderr": 1.307,
                "cost_per_test": 0.00028,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "together/meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo": {
                "accuracy": 21.504,
                "latency": 2.458,
                "stderr": 1.175,
                "cost_per_test": 0.000107,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "poolside/laguna-m.1": {
                "accuracy": 1.472,
                "latency": 212.643,
                "stderr": 0.344,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            }
        },
        "stepwise_reasoning": {
            "meta/muse_spark_1_2": {
                "accuracy": 92.723,
                "latency": 48.313,
                "stderr": 0.743,
                "cost_per_test": 0.014199,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "meta/muse_spark_1_1": {
                "accuracy": 92.559,
                "latency": 32.816,
                "stderr": 0.75,
                "cost_per_test": 0.014921,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "meta/muse_spark": {
                "accuracy": 89.779,
                "latency": 57.522,
                "stderr": 0.866,
                "cost_per_test": 0.001922,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 88.389,
                "latency": 127.879,
                "stderr": 0.916,
                "cost_per_test": 0.127973,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-5": {
                "accuracy": 88.307,
                "latency": 58.339,
                "stderr": 0.919,
                "cost_per_test": 0.114274,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-fable-5": {
                "accuracy": 88.062,
                "latency": 56.865,
                "stderr": 0.927,
                "cost_per_test": 0.220133,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-3": {
                "accuracy": 87.899,
                "latency": 13.058,
                "stderr": 0.933,
                "cost_per_test": 0.017475,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-fable-5-1": {
                "accuracy": 87.817,
                "latency": 85.564,
                "stderr": 0.935,
                "cost_per_test": 0.346549,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-sonnet-5": {
                "accuracy": 87.244,
                "latency": 202.488,
                "stderr": 0.954,
                "cost_per_test": 0.301619,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "zai/glm-5.3-flash": {
                "accuracy": 87.163,
                "latency": 53.06,
                "stderr": 0.957,
                "cost_per_test": 0.001505,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 87.163,
                "latency": 79.254,
                "stderr": 0.957,
                "cost_per_test": 0.071368,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "thinkingmachines/inkling-small": {
                "accuracy": 87.163,
                "latency": 241.592,
                "stderr": 0.957,
                "cost_per_test": 0.017501,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 262000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "alibaba/qwen3.7-max": {
                "accuracy": 87.081,
                "latency": 57.677,
                "stderr": 0.959,
                "cost_per_test": 0.028122,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "thinkingmachines/inkling": {
                "accuracy": 87.081,
                "latency": 197.78,
                "stderr": 0.959,
                "cost_per_test": 0.103224,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "alibaba/qwen3.8-max": {
                "accuracy": 86.999,
                "latency": 265.751,
                "stderr": 0.962,
                "cost_per_test": 0.07645,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 86.917,
                "latency": 90.464,
                "stderr": 0.964,
                "cost_per_test": 0.014221,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "google/gemini-3.5-flash": {
                "accuracy": 86.836,
                "latency": 19.694,
                "stderr": 0.967,
                "cost_per_test": 0.038136,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-4-7": {
                "accuracy": 86.754,
                "latency": 62.071,
                "stderr": 0.969,
                "cost_per_test": 0.124592,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-4-8": {
                "accuracy": 86.754,
                "latency": 70.33,
                "stderr": 0.969,
                "cost_per_test": 0.159139,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 86.59,
                "latency": 267.484,
                "stderr": 0.974,
                "cost_per_test": 0.029274,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "openai/gpt-5.6-terra": {
                "accuracy": 86.509,
                "latency": 37.28,
                "stderr": 0.977,
                "cost_per_test": 0.035909,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3.8-flash": {
                "accuracy": 86.427,
                "latency": 8.759,
                "stderr": 0.979,
                "cost_per_test": 0.023535,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-4.1-2025-04-14": {
                "accuracy": 86.263,
                "latency": 7.89,
                "stderr": 0.984,
                "cost_per_test": 0.006214,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3.6-flash": {
                "accuracy": 86.182,
                "latency": 12.335,
                "stderr": 0.987,
                "cost_per_test": 0.020385,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "kimi/kimi-k3": {
                "accuracy": 86.1,
                "latency": 144.708,
                "stderr": 0.989,
                "cost_per_test": 0.078156,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "google/gemini-3.7-flash": {
                "accuracy": 85.854,
                "latency": 7.832,
                "stderr": 0.997,
                "cost_per_test": 0.018681,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 85.854,
                "latency": 47.448,
                "stderr": 0.997,
                "cost_per_test": 0.218987,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.6-luna": {
                "accuracy": 85.773,
                "latency": 76.52,
                "stderr": 0.999,
                "cost_per_test": 0.012217,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 85.691,
                "latency": 37.434,
                "stderr": 1.001,
                "cost_per_test": 0.011528,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101": {
                "accuracy": 85.527,
                "latency": 14.876,
                "stderr": 1.006,
                "cost_per_test": 0.058665,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3-max-preview": {
                "accuracy": 85.527,
                "latency": 21.405,
                "stderr": 1.006,
                "cost_per_test": 0.00671,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 85.446,
                "latency": 12.001,
                "stderr": 1.008,
                "cost_per_test": 0.005764,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "minimax/MiniMax-M3": {
                "accuracy": 85.446,
                "latency": 94.036,
                "stderr": 1.008,
                "cost_per_test": 0.00728,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 512000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "alibaba/qwen3-max": {
                "accuracy": 85.282,
                "latency": 29.296,
                "stderr": 1.013,
                "cost_per_test": 0.00654,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 85.282,
                "latency": 70.634,
                "stderr": 1.013,
                "cost_per_test": 0.01145,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 85.2,
                "latency": 48.391,
                "stderr": 1.015,
                "cost_per_test": 0.040814,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 85.2,
                "latency": 60.479,
                "stderr": 1.015,
                "cost_per_test": 0.066435,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/o4-mini-2025-04-16": {
                "accuracy": 85.119,
                "latency": 15.447,
                "stderr": 1.018,
                "cost_per_test": 0.007057,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/o3-2025-04-16": {
                "accuracy": 85.119,
                "latency": 24.024,
                "stderr": 1.018,
                "cost_per_test": 0.012997,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 100000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 85.016,
                "latency": 41.931,
                "stderr": 1.019,
                "cost_per_test": 0.03149,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 84.873,
                "latency": 13.394,
                "stderr": 1.025,
                "cost_per_test": 0.017152,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 84.873,
                "latency": 44.176,
                "stderr": 1.025,
                "cost_per_test": 0.007193,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-3-7-sonnet-20250219-thinking": {
                "accuracy": 84.791,
                "latency": 43.314,
                "stderr": 1.027,
                "cost_per_test": 0.050764,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 18384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 84.628,
                "latency": 30.758,
                "stderr": 1.031,
                "cost_per_test": 0.001333,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "zai/glm-5.2": {
                "accuracy": 84.628,
                "latency": 64.942,
                "stderr": 1.031,
                "cost_per_test": 0.015431,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "mistralai/mistral-large-2512": {
                "accuracy": 84.546,
                "latency": 29.531,
                "stderr": 1.034,
                "cost_per_test": 0.001601,
                "temperature": 0.065,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "nvidia/nemotron-3-ultra-550b-a55b": {
                "accuracy": 84.546,
                "latency": 36.485,
                "stderr": 1.034,
                "cost_per_test": null,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Nvidia",
                "harness": null
            },
            "openai/gpt-5.6-sol": {
                "accuracy": 84.546,
                "latency": 62.034,
                "stderr": 1.034,
                "cost_per_test": 0.138418,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025": {
                "accuracy": 84.464,
                "latency": 6.267,
                "stderr": 1.036,
                "cost_per_test": 0.003098,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "xiaomi/mimo-v2.5-pro": {
                "accuracy": 84.464,
                "latency": 62.829,
                "stderr": 1.036,
                "cost_per_test": 0.002547,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 84.464,
                "latency": 95.83,
                "stderr": 1.036,
                "cost_per_test": 0.06042,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-4-1-20250805-thinking": {
                "accuracy": 84.383,
                "latency": 26.637,
                "stderr": 1.038,
                "cost_per_test": 0.093482,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-4o-2024-11-20": {
                "accuracy": 84.219,
                "latency": 5.855,
                "stderr": 1.042,
                "cost_per_test": 0.005806,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-2.5-pro-exp-03-25": {
                "accuracy": 84.219,
                "latency": 21.775,
                "stderr": 1.042,
                "cost_per_test": 0.010628,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025-thinking": {
                "accuracy": 84.137,
                "latency": 10.635,
                "stderr": 1.045,
                "cost_per_test": 0.003099,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/o1-2024-12-17": {
                "accuracy": 84.137,
                "latency": 19.477,
                "stderr": 1.045,
                "cost_per_test": 0.12173,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 84.056,
                "latency": 24.404,
                "stderr": 1.047,
                "cost_per_test": 0.034377,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "zai/glm-4.5": {
                "accuracy": 84.056,
                "latency": 129.388,
                "stderr": 1.047,
                "cost_per_test": 0.009559,
                "temperature": 0.6,
                "top_p": 1.0,
                "max_output_tokens": 81920,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "grok/grok-3-mini-fast-low-reasoning": {
                "accuracy": 83.974,
                "latency": 11.728,
                "stderr": 1.049,
                "cost_per_test": 0.00112,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "low",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 83.974,
                "latency": 78.668,
                "stderr": 1.049,
                "cost_per_test": 0.053873,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "fireworks/deepseek-r1": {
                "accuracy": 83.81,
                "latency": 152.958,
                "stderr": 1.053,
                "cost_per_test": 0.012239,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "fireworks/gpt-oss-120b": {
                "accuracy": 83.729,
                "latency": 47.661,
                "stderr": 1.055,
                "cost_per_test": 0.002958,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "kimi/kimi-k2-thinking": {
                "accuracy": 83.729,
                "latency": 61.408,
                "stderr": 1.055,
                "cost_per_test": 0.00913,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "grok/grok-4-fast-non-reasoning": {
                "accuracy": 83.603,
                "latency": 4.25,
                "stderr": 0.943,
                "cost_per_test": 0.000571,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "alibaba/qwen3.5-flash": {
                "accuracy": 83.565,
                "latency": 73.003,
                "stderr": 1.06,
                "cost_per_test": 0.003099,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 83.565,
                "latency": 189.695,
                "stderr": 1.06,
                "cost_per_test": 0.033187,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "google/gemini-3.5-flash-lite": {
                "accuracy": 83.483,
                "latency": 8.815,
                "stderr": 1.062,
                "cost_per_test": 0.00729,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514-thinking": {
                "accuracy": 83.483,
                "latency": 32.385,
                "stderr": 1.062,
                "cost_per_test": 0.029421,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "deepseek/deepseek-v4-pro-0813": {
                "accuracy": 83.483,
                "latency": 226.296,
                "stderr": 1.062,
                "cost_per_test": 0.045816,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "anthropic/claude-opus-4-20250514": {
                "accuracy": 83.32,
                "latency": 15.23,
                "stderr": 1.066,
                "cost_per_test": 0.03694,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3.6-27b": {
                "accuracy": 83.32,
                "latency": 147.778,
                "stderr": 1.066,
                "cost_per_test": 0.030256,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "fireworks/deepseek-v3-0324": {
                "accuracy": 83.238,
                "latency": 34.009,
                "stderr": 1.068,
                "cost_per_test": 0.000652,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 83.156,
                "latency": 7.644,
                "stderr": 1.07,
                "cost_per_test": 0.00096,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-4-fast-reasoning": {
                "accuracy": 83.15,
                "latency": 10.184,
                "stderr": 0.953,
                "cost_per_test": 0.001268,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-opus-4-1-20250805": {
                "accuracy": 83.074,
                "latency": 32.602,
                "stderr": 1.072,
                "cost_per_test": 0.039842,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/gpt-4.1-mini-2025-04-14": {
                "accuracy": 82.993,
                "latency": 5.921,
                "stderr": 1.074,
                "cost_per_test": 0.001233,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 82.993,
                "latency": 78.717,
                "stderr": 1.074,
                "cost_per_test": 0.157344,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-3-7-sonnet-20250219": {
                "accuracy": 82.911,
                "latency": 6.28,
                "stderr": 1.076,
                "cost_per_test": 0.006427,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "xiaomi/mimo-v2.5": {
                "accuracy": 82.911,
                "latency": 26.3,
                "stderr": 1.076,
                "cost_per_test": 0.000729,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-04-17": {
                "accuracy": 82.747,
                "latency": 5.388,
                "stderr": 1.08,
                "cost_per_test": 0.00261,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "zai/glm-5.3": {
                "accuracy": 82.747,
                "latency": 98.323,
                "stderr": 1.08,
                "cost_per_test": 0.032,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "fireworks/qwen3-235b-a22b": {
                "accuracy": 82.257,
                "latency": 92.742,
                "stderr": 1.092,
                "cost_per_test": 0.004644,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-04-17-thinking": {
                "accuracy": 82.175,
                "latency": 15.355,
                "stderr": 1.094,
                "cost_per_test": 0.007461,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-3-mini-fast-high-reasoning": {
                "accuracy": 82.093,
                "latency": 25.447,
                "stderr": 1.096,
                "cost_per_test": 0.002142,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "mistralai/mistral-medium-2505": {
                "accuracy": 81.848,
                "latency": 11.052,
                "stderr": 1.102,
                "cost_per_test": 0.001422,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "together/moonshotai/Kimi-K2-Instruct": {
                "accuracy": 81.603,
                "latency": 12.834,
                "stderr": 1.108,
                "cost_per_test": 0.001543,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "alibaba/qwen3.8-27b": {
                "accuracy": 81.603,
                "latency": 94.491,
                "stderr": 1.108,
                "cost_per_test": 0.046465,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/gpt-4o-2024-08-06": {
                "accuracy": 81.439,
                "latency": 9.543,
                "stderr": 1.112,
                "cost_per_test": 0.004417,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "ant/ling-3.0-flash-2607": {
                "accuracy": 81.357,
                "latency": 7.54,
                "stderr": 1.114,
                "cost_per_test": 0.000664,
                "temperature": 0.6,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514": {
                "accuracy": 81.357,
                "latency": 11.847,
                "stderr": 1.114,
                "cost_per_test": 0.0076,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-2.0-flash-thinking-exp-01-21": {
                "accuracy": 81.276,
                "latency": 12.807,
                "stderr": 1.116,
                "cost_per_test": 0.000827,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "zai/glm-4.6": {
                "accuracy": 81.141,
                "latency": 64.149,
                "stderr": 0.996,
                "cost_per_test": 0.00928,
                "temperature": 0.6,
                "top_p": 1.0,
                "max_output_tokens": 122880,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "grok/grok-4.3": {
                "accuracy": 81.03,
                "latency": 391.972,
                "stderr": 1.121,
                "cost_per_test": 0.013656,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "deepseek/deepseek-v4-flash-0731": {
                "accuracy": 80.867,
                "latency": 72.203,
                "stderr": 1.125,
                "cost_per_test": 0.009755,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "minimax/MiniMax-M2.5": {
                "accuracy": 80.867,
                "latency": 73.243,
                "stderr": 1.125,
                "cost_per_test": 0.008629,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "zai/glm-5-thinking": {
                "accuracy": 80.867,
                "latency": 149.298,
                "stderr": 1.125,
                "cost_per_test": 0.021913,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "zai/glm-5.1": {
                "accuracy": 80.851,
                "latency": 84.072,
                "stderr": 1.126,
                "cost_per_test": 0.016958,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "grok/grok-4.5": {
                "accuracy": 80.785,
                "latency": 195.458,
                "stderr": 1.127,
                "cost_per_test": 0.019211,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-3-5-sonnet-20241022": {
                "accuracy": 80.458,
                "latency": 5.943,
                "stderr": 1.134,
                "cost_per_test": 0.006006,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-4.6": {
                "accuracy": 80.049,
                "latency": 47.57,
                "stderr": 1.143,
                "cost_per_test": 0.017298,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-5-nano-2025-08-07": {
                "accuracy": 80.049,
                "latency": 69.944,
                "stderr": 1.143,
                "cost_per_test": 0.004904,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": "medium",
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 79.804,
                "latency": 47.524,
                "stderr": 1.148,
                "cost_per_test": 0.02743,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-4-1-fast-non-reasoning": {
                "accuracy": 79.64,
                "latency": 9.277,
                "stderr": 1.151,
                "cost_per_test": 0.000372,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/o3-mini-2025-01-31": {
                "accuracy": 79.477,
                "latency": 74.312,
                "stderr": 1.155,
                "cost_per_test": 0.018142,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-2.0-flash-exp": {
                "accuracy": 79.395,
                "latency": 7.486,
                "stderr": 1.157,
                "cost_per_test": 0.000302,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.4-mini-2026-03-17": {
                "accuracy": 79.395,
                "latency": 44.578,
                "stderr": 1.157,
                "cost_per_test": 0.00677,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-2-1212": {
                "accuracy": 79.231,
                "latency": 10.621,
                "stderr": 1.16,
                "cost_per_test": 0.006894,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 0,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "mistralai/mistral-medium-3.5": {
                "accuracy": 79.231,
                "latency": 102.392,
                "stderr": 1.16,
                "cost_per_test": 0.079101,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 80000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "fireworks/llama4-maverick-instruct-basic": {
                "accuracy": 79.15,
                "latency": 25.143,
                "stderr": 1.162,
                "cost_per_test": 0.000527,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "google/gemini-2.0-pro-exp-02-05": {
                "accuracy": 78.986,
                "latency": 8.989,
                "stderr": 1.165,
                "cost_per_test": 0.004012,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 78.986,
                "latency": 28.694,
                "stderr": 1.165,
                "cost_per_test": 0.003474,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 196608,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "fireworks/deepseek-v3": {
                "accuracy": 78.741,
                "latency": 27.696,
                "stderr": 1.17,
                "cost_per_test": 0.00056,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025": {
                "accuracy": 78.496,
                "latency": 3.597,
                "stderr": 1.175,
                "cost_per_test": 0.000477,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "fireworks/deepseek-v3p2-thinking": {
                "accuracy": 78.414,
                "latency": 98.919,
                "stderr": 1.176,
                "cost_per_test": 0.002376,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 20480,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "minimax/MiniMax-M2.1": {
                "accuracy": 78.25,
                "latency": 113.258,
                "stderr": 1.18,
                "cost_per_test": 0.009137,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025-thinking": {
                "accuracy": 77.433,
                "latency": 3.94,
                "stderr": 1.195,
                "cost_per_test": 0.000509,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65535,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 77.351,
                "latency": 15.898,
                "stderr": 1.197,
                "cost_per_test": 0.000638,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-4.7": {
                "accuracy": 77.269,
                "latency": 187.181,
                "stderr": 1.198,
                "cost_per_test": 0.013873,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "google/gemini-2.0-flash-001": {
                "accuracy": 77.024,
                "latency": 5.702,
                "stderr": 1.203,
                "cost_per_test": 0.000296,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "fireworks/gpt-oss-20b": {
                "accuracy": 76.697,
                "latency": 49.004,
                "stderr": 1.209,
                "cost_per_test": 0.002254,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "together/langston/nim/nvidia/llama-3.3-nemotron-super-49b-v1-42e84561-thinking": {
                "accuracy": 75.388,
                "latency": 32.67,
                "stderr": 1.232,
                "cost_per_test": null,
                "temperature": 0.6,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "ai21labs/jamba-large-1.6": {
                "accuracy": 74.244,
                "latency": 16.439,
                "stderr": 1.25,
                "cost_per_test": 0.005162,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "mistralai/mistral-large-2411": {
                "accuracy": 73.917,
                "latency": 12.019,
                "stderr": 1.256,
                "cost_per_test": 0.004079,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "cohere/command-a-03-2025": {
                "accuracy": 73.753,
                "latency": 11.214,
                "stderr": 1.258,
                "cost_per_test": 0.006484,
                "temperature": 0.3,
                "top_p": null,
                "max_output_tokens": 8000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "openai/gpt-4.1-nano-2025-04-14": {
                "accuracy": 72.935,
                "latency": 2.998,
                "stderr": 1.27,
                "cost_per_test": 0.000291,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "together/meta-llama/Meta-Llama-3.1-405B-Instruct-Turbo": {
                "accuracy": 72.445,
                "latency": 23.551,
                "stderr": 1.278,
                "cost_per_test": 0.001904,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "together/langston/nim/nvidia/llama-3.3-nemotron-super-49b-v1-42e84561": {
                "accuracy": 72.363,
                "latency": 18.308,
                "stderr": 1.279,
                "cost_per_test": null,
                "temperature": 0.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "google/gemini-1.5-pro-002": {
                "accuracy": 71.627,
                "latency": 9.86,
                "stderr": 1.289,
                "cost_per_test": 0.002545,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "together/meta-llama/Llama-3.3-70B-Instruct-Turbo": {
                "accuracy": 71.545,
                "latency": 3.836,
                "stderr": 1.29,
                "cost_per_test": 0.000532,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "openai/gpt-4o-mini-2024-07-18": {
                "accuracy": 71.545,
                "latency": 8.899,
                "stderr": 1.29,
                "cost_per_test": 0.000323,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "poolside/laguna-xs.2": {
                "accuracy": 71.218,
                "latency": 27.449,
                "stderr": 1.295,
                "cost_per_test": 0.000305,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "mistralai/magistral-medium-2509": {
                "accuracy": 71.055,
                "latency": 46.254,
                "stderr": 1.297,
                "cost_per_test": 0.025214,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "mistralai/magistral-small-2509": {
                "accuracy": 70.564,
                "latency": 16.441,
                "stderr": 1.303,
                "cost_per_test": 0.003909,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "mistralai/mistral-small-2503": {
                "accuracy": 70.074,
                "latency": 7.952,
                "stderr": 1.309,
                "cost_per_test": 0.000184,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "ai21labs/jamba-1.5-large": {
                "accuracy": 69.583,
                "latency": 20.028,
                "stderr": 1.316,
                "cost_per_test": 0.005384,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "together/meta-llama/Meta-Llama-3.1-70B-Instruct-Turbo": {
                "accuracy": 68.193,
                "latency": 4.327,
                "stderr": 1.332,
                "cost_per_test": 0.000488,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "anthropic/claude-3-5-haiku-20241022": {
                "accuracy": 68.193,
                "latency": 4.987,
                "stderr": 1.332,
                "cost_per_test": 0.001317,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-4-0709": {
                "accuracy": 68.111,
                "latency": 159.421,
                "stderr": 1.333,
                "cost_per_test": 0.077989,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "together/meta-llama/Llama-4-Scout-17B-16E-Instruct": {
                "accuracy": 67.375,
                "latency": 7.042,
                "stderr": 1.341,
                "cost_per_test": 0.000378,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 16384,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "google/gemini-1.5-flash-002": {
                "accuracy": 60.262,
                "latency": 3.701,
                "stderr": 1.399,
                "cost_per_test": 0.000155,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 8192,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "mistralai/mistral-small-2402": {
                "accuracy": 60.016,
                "latency": 9.018,
                "stderr": 1.401,
                "cost_per_test": 0.000355,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 256000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "ai21labs/jamba-mini-1.6": {
                "accuracy": 58.872,
                "latency": 4.337,
                "stderr": 1.407,
                "cost_per_test": 0.00026,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "ai21labs/jamba-1.5-mini": {
                "accuracy": 53.966,
                "latency": 4.78,
                "stderr": 1.425,
                "cost_per_test": 0.00028,
                "temperature": 0.4,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "AI21 Labs",
                "harness": null
            },
            "together/meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo": {
                "accuracy": 43.173,
                "latency": 2.458,
                "stderr": 1.416,
                "cost_per_test": 0.000107,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 4096,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "poolside/laguna-m.1": {
                "accuracy": 1.799,
                "latency": 212.643,
                "stderr": 0.38,
                "cost_per_test": null,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 32768,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            }
        }
    }
}