{
    "metadata": {
        "benchmark": "MedScribe",
        "slug": "medscribe",
        "description": "Can models support doctors with their administrative work?",
        "benchmark_id": "medscribe",
        "family": "medscribe",
        "version": "1",
        "updated": "2026-10-01",
        "dataset_type": "private",
        "industry": "healthcare",
        "tasks": {
            "overall": "Overall"
        },
        "models": [
            "alibaba/qwen3-max-2026-01-23",
            "alibaba/qwen3-vl-plus-2025-09-23",
            "alibaba/qwen3.5-flash",
            "alibaba/qwen3.6-plus",
            "alibaba/qwen3.7-max",
            "alibaba/qwen3.8-27b",
            "alibaba/qwen3.8-max",
            "ant/ling-3.0-flash-2607",
            "ant/ling-3.0-flash-af-rc3",
            "anthropic/claude-fable-5",
            "anthropic/claude-fable-5-1",
            "anthropic/claude-haiku-4-5-20251001-thinking",
            "anthropic/claude-opus-4-1-20250805",
            "anthropic/claude-opus-4-1-20250805-thinking",
            "anthropic/claude-opus-4-5-20251101",
            "anthropic/claude-opus-4-5-20251101-thinking",
            "anthropic/claude-opus-4-6",
            "anthropic/claude-opus-4-6-thinking",
            "anthropic/claude-opus-4-7",
            "anthropic/claude-opus-4-8",
            "anthropic/claude-opus-5",
            "anthropic/claude-opus-5-5",
            "anthropic/claude-sonnet-4-20250514",
            "anthropic/claude-sonnet-4-20250514-thinking",
            "anthropic/claude-sonnet-4-5-20250929",
            "anthropic/claude-sonnet-4-5-20250929-thinking",
            "anthropic/claude-sonnet-5",
            "anthropic/claude-sonnet-5-5",
            "cohere/command-a-plus-05-2026",
            "deepseek/deepseek-v4-flash-0731",
            "deepseek/deepseek-v4-pro",
            "deepseek/deepseek-v4-pro-0813",
            "deepseek/deepseek-v4.1-flash",
            "fireworks/llama4-maverick-instruct-basic",
            "fireworks/nemotron-lightning-3p5-30b-a3b",
            "google/gemini-2.5-flash",
            "google/gemini-2.5-flash-lite",
            "google/gemini-2.5-flash-lite-preview-09-2025",
            "google/gemini-2.5-flash-lite-preview-09-2025-thinking",
            "google/gemini-2.5-flash-preview-09-2025",
            "google/gemini-2.5-flash-preview-09-2025-thinking",
            "google/gemini-2.5-flash-thinking",
            "google/gemini-2.5-pro",
            "google/gemini-3-flash-preview",
            "google/gemini-3-pro-preview",
            "google/gemini-3.1-flash-lite-preview",
            "google/gemini-3.1-pro-preview",
            "google/gemini-3.5-flash",
            "google/gemini-3.5-flash-lite",
            "google/gemini-3.6-flash",
            "google/gemini-3.7-flash",
            "google/gemini-3.8-flash",
            "google/gemini-4-argon",
            "grok/grok-4-0709",
            "grok/grok-4-1-fast-non-reasoning",
            "grok/grok-4-1-fast-reasoning",
            "grok/grok-4-fast-non-reasoning",
            "grok/grok-4-fast-reasoning",
            "grok/grok-4.20-0309-reasoning",
            "grok/grok-4.3",
            "grok/grok-4.5",
            "grok/grok-4.6",
            "grok/grok-4.7",
            "inception/mercury-2.5",
            "kimi/kimi-k2.5-thinking",
            "kimi/kimi-k2.6",
            "kimi/kimi-k3",
            "meta/muse_spark",
            "meta/muse_spark_1_1",
            "meta/muse_spark_1_2",
            "minimax/MiniMax-M2.1",
            "minimax/MiniMax-M2.7",
            "minimax/MiniMax-M3",
            "mistralai/mistral-medium-3.5",
            "openai/gpt-5-2025-08-07",
            "openai/gpt-5-mini-2025-08-07",
            "openai/gpt-5-nano-2025-08-07",
            "openai/gpt-5.1-2025-11-13",
            "openai/gpt-5.2-2025-12-11",
            "openai/gpt-5.4-2026-03-05",
            "openai/gpt-5.4-nano-2026-03-17",
            "openai/gpt-5.5",
            "openai/gpt-5.6-luna",
            "openai/gpt-5.6-sol",
            "openai/gpt-5.6-terra",
            "openai/gpt-6-astra",
            "openai/gpt-6-luna",
            "openai/gpt-6-sol",
            "openai/gpt-6.1-sol",
            "openai/o3-2025-04-16",
            "openai/o4-mini-2025-04-16",
            "poolside/laguna-m.1",
            "poolside/laguna-xs.2",
            "tencent/hy4-preview",
            "thinkingmachines/inkling",
            "thinkingmachines/inkling-small",
            "together/meta-llama/Llama-4-Scout-17B-16E-Instruct",
            "xiaomi/mimo-v2.5",
            "xiaomi/mimo-v2.5-pro",
            "xiaomi/mimo-v2.6-flash",
            "xiaomi/mimo-v2.6-pro",
            "zai/glm-4.7",
            "zai/glm-5.1",
            "zai/glm-5.2",
            "zai/glm-5.3",
            "zai/glm-5.3-flash"
        ],
        "partners": [],
        "showBadge": false,
        "visible": true,
        "use_cost_per_test": false,
        "runner": "platform",
        "mode": "one-shot",
        "archived": false,
        "partner": false,
        "total_models": 106
    },
    "tasks": {
        "overall": {
            "anthropic/claude-opus-5-5": {
                "accuracy": 91.43,
                "latency": 441.582,
                "stderr": 1.932,
                "cost_per_test": 1.154156,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-fable-5-1": {
                "accuracy": 91.294,
                "latency": 188.363,
                "stderr": 1.953,
                "cost_per_test": 0.9635,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-sonnet-5-5": {
                "accuracy": 91.101,
                "latency": 306.811,
                "stderr": 1.96,
                "cost_per_test": 0.508604,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "anthropic/claude-opus-5": {
                "accuracy": 90.985,
                "latency": 76.563,
                "stderr": 1.916,
                "cost_per_test": 0.236275,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "meta/muse_spark_1_2": {
                "accuracy": 90.062,
                "latency": 61.458,
                "stderr": 1.958,
                "cost_per_test": 0.037779,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "grok/grok-4.7": {
                "accuracy": 89.377,
                "latency": 153.053,
                "stderr": 1.886,
                "cost_per_test": 0.074505,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": null,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "zai/glm-5.3-flash": {
                "accuracy": 88.936,
                "latency": 86.98,
                "stderr": 1.907,
                "cost_per_test": 0.002235,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "meta/muse_spark_1_1": {
                "accuracy": 88.888,
                "latency": 63.335,
                "stderr": 1.95,
                "cost_per_test": 0.034628,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "zai/glm-5.3": {
                "accuracy": 88.81,
                "latency": 121.931,
                "stderr": 1.999,
                "cost_per_test": 0.057236,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "anthropic/claude-fable-5": {
                "accuracy": 88.522,
                "latency": 119.475,
                "stderr": 1.945,
                "cost_per_test": 0.583239,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "xiaomi/mimo-v2.6-pro": {
                "accuracy": 88.307,
                "latency": 173.345,
                "stderr": 1.938,
                "cost_per_test": 0.00988,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 88.09,
                "latency": 77.983,
                "stderr": 1.942,
                "cost_per_test": 0.096508,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "kimi/kimi-k3": {
                "accuracy": 88.048,
                "latency": 38.241,
                "stderr": 1.976,
                "cost_per_test": 0.110262,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "openai/gpt-6-astra": {
                "accuracy": 87.908,
                "latency": 153.728,
                "stderr": 1.938,
                "cost_per_test": 0.581991,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-4-argon": {
                "accuracy": 87.427,
                "latency": 87.439,
                "stderr": 1.964,
                "cost_per_test": 0.147175,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 262144,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "minimax/MiniMax-M3": {
                "accuracy": 87.252,
                "latency": 123.967,
                "stderr": 1.957,
                "cost_per_test": 0.013748,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "grok/grok-4.5": {
                "accuracy": 86.884,
                "latency": 704.513,
                "stderr": 1.944,
                "cost_per_test": 0.033208,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-5.5": {
                "accuracy": 86.868,
                "latency": 132.761,
                "stderr": 1.932,
                "cost_per_test": 0.142988,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-4-6": {
                "accuracy": 86.738,
                "latency": 54.318,
                "stderr": 1.942,
                "cost_per_test": 0.115121,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "grok/grok-4.6": {
                "accuracy": 86.534,
                "latency": 70.944,
                "stderr": 1.956,
                "cost_per_test": 0.03719,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "openai/gpt-6.1-sol": {
                "accuracy": 86.453,
                "latency": 169.619,
                "stderr": 1.898,
                "cost_per_test": 0.100902,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 86.13,
                "latency": 127.564,
                "stderr": 1.944,
                "cost_per_test": 0.224735,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "meta/muse_spark": {
                "accuracy": 85.902,
                "latency": 191.473,
                "stderr": 1.847,
                "cost_per_test": 0.007681,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Meta",
                "harness": null
            },
            "anthropic/claude-opus-4-8": {
                "accuracy": 85.755,
                "latency": 82.364,
                "stderr": 1.928,
                "cost_per_test": 0.259121,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "deepseek/deepseek-v4.1-flash": {
                "accuracy": 85.5,
                "latency": 53.826,
                "stderr": 1.918,
                "cost_per_test": 0.015397,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 384000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "thinkingmachines/inkling": {
                "accuracy": 85.405,
                "latency": 244.954,
                "stderr": 1.844,
                "cost_per_test": 0.165561,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 85.321,
                "latency": 72.112,
                "stderr": 1.896,
                "cost_per_test": 0.410224,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "xiaomi/mimo-v2.6-flash": {
                "accuracy": 85.275,
                "latency": 74.481,
                "stderr": 1.986,
                "cost_per_test": 0.002184,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "openai/gpt-5.6-sol": {
                "accuracy": 85.233,
                "latency": 94.404,
                "stderr": 1.973,
                "cost_per_test": 0.276691,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 85.23,
                "latency": 66.197,
                "stderr": 1.899,
                "cost_per_test": 0.042375,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "alibaba/qwen3.8-max": {
                "accuracy": 84.947,
                "latency": 251.26,
                "stderr": 1.999,
                "cost_per_test": 0.089616,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "anthropic/claude-sonnet-4-5-20250929": {
                "accuracy": 84.515,
                "latency": 44.419,
                "stderr": 1.929,
                "cost_per_test": 0.054649,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.8-flash": {
                "accuracy": 84.496,
                "latency": 21.145,
                "stderr": 1.943,
                "cost_per_test": 0.025238,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.6-luna": {
                "accuracy": 84.391,
                "latency": 116.886,
                "stderr": 2.585,
                "cost_per_test": 0.022813,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 84.387,
                "latency": 126.205,
                "stderr": 1.856,
                "cost_per_test": 0.115422,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "thinkingmachines/inkling-small": {
                "accuracy": 84.112,
                "latency": 229.993,
                "stderr": 1.87,
                "cost_per_test": 0.019018,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": "0.99",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Thinkingmachines",
                "harness": null
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 84.101,
                "latency": 67.924,
                "stderr": 1.873,
                "cost_per_test": 0.082281,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.7-flash": {
                "accuracy": 83.942,
                "latency": 16.818,
                "stderr": 2.004,
                "cost_per_test": 0.058736,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3.8-27b": {
                "accuracy": 83.849,
                "latency": 89.705,
                "stderr": 1.977,
                "cost_per_test": 0.046732,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "xiaomi/mimo-v2.5-pro": {
                "accuracy": 83.73,
                "latency": 90.71,
                "stderr": 2.063,
                "cost_per_test": 0.00603,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "openai/gpt-6-luna": {
                "accuracy": 83.71,
                "latency": 169.606,
                "stderr": 1.949,
                "cost_per_test": 0.009562,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 83.65,
                "latency": 184.536,
                "stderr": 1.936,
                "cost_per_test": 0.1015,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "tencent/hy4-preview": {
                "accuracy": 83.6,
                "latency": 349.623,
                "stderr": 2.065,
                "cost_per_test": 0.053243,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Tencent",
                "harness": null
            },
            "zai/glm-5.2": {
                "accuracy": 83.534,
                "latency": 138.372,
                "stderr": 2.002,
                "cost_per_test": 0.044912,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "anthropic/claude-opus-4-5-20251101": {
                "accuracy": 83.246,
                "latency": 43.305,
                "stderr": 1.926,
                "cost_per_test": 0.281674,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "high",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-2.5-flash-thinking": {
                "accuracy": 82.983,
                "latency": 22.529,
                "stderr": 1.908,
                "cost_per_test": 0.014824,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-4-7": {
                "accuracy": 82.953,
                "latency": 67.496,
                "stderr": 1.977,
                "cost_per_test": 0.177841,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-2.5-flash": {
                "accuracy": 82.869,
                "latency": 22.786,
                "stderr": 1.909,
                "cost_per_test": 0.014869,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.6-terra": {
                "accuracy": 82.869,
                "latency": 35.526,
                "stderr": 1.948,
                "cost_per_test": 0.062588,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "openai/gpt-6-sol": {
                "accuracy": 82.034,
                "latency": 81.135,
                "stderr": 1.942,
                "cost_per_test": 0.083138,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4-fast-reasoning": {
                "accuracy": 81.632,
                "latency": 12.423,
                "stderr": 2.137,
                "cost_per_test": 0.002535,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "ant/ling-3.0-flash-2607": {
                "accuracy": 80.904,
                "latency": 12.121,
                "stderr": 2.039,
                "cost_per_test": 0.001358,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "minimax/MiniMax-M2.1": {
                "accuracy": 80.777,
                "latency": 53.161,
                "stderr": 1.831,
                "cost_per_test": 0.005087,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 80.577,
                "latency": 275.717,
                "stderr": 1.924,
                "cost_per_test": 0.033478,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "deepseek/deepseek-v4-flash-0731": {
                "accuracy": 80.363,
                "latency": 77.552,
                "stderr": 1.973,
                "cost_per_test": 0.014247,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "deepseek/deepseek-v4-pro-0813": {
                "accuracy": 80.174,
                "latency": 155.859,
                "stderr": 2.004,
                "cost_per_test": 0.041127,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "minimax/MiniMax-M2.7": {
                "accuracy": 79.867,
                "latency": 27.25,
                "stderr": 1.86,
                "cost_per_test": 0.005124,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "MiniMax",
                "harness": null
            },
            "grok/grok-4-fast-non-reasoning": {
                "accuracy": 79.722,
                "latency": 8.482,
                "stderr": 1.871,
                "cost_per_test": 0.002056,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-3.6-flash": {
                "accuracy": 79.661,
                "latency": 33.552,
                "stderr": 1.861,
                "cost_per_test": 0.073178,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3.7-max": {
                "accuracy": 79.396,
                "latency": 108.046,
                "stderr": 1.907,
                "cost_per_test": 0.06907,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 78.732,
                "latency": 39.292,
                "stderr": 1.866,
                "cost_per_test": 0.002387,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025-thinking": {
                "accuracy": 78.497,
                "latency": 31.385,
                "stderr": 1.993,
                "cost_per_test": 0.014526,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-4-0709": {
                "accuracy": 78.152,
                "latency": 74.054,
                "stderr": 2.084,
                "cost_per_test": 0.063955,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "kimi/kimi-k2.6": {
                "accuracy": 78.149,
                "latency": 494.142,
                "stderr": 1.792,
                "cost_per_test": 0.055962,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "google/gemini-2.5-flash-preview-09-2025": {
                "accuracy": 77.946,
                "latency": 22.893,
                "stderr": 1.923,
                "cost_per_test": 0.014385,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5.4-2026-03-05": {
                "accuracy": 77.549,
                "latency": 282.914,
                "stderr": 3.316,
                "cost_per_test": 0.639282,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "grok/grok-4-1-fast-non-reasoning": {
                "accuracy": 77.464,
                "latency": 19.865,
                "stderr": 2.04,
                "cost_per_test": 0.001782,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "alibaba/qwen3-vl-plus-2025-09-23": {
                "accuracy": 77.129,
                "latency": 71.346,
                "stderr": 1.916,
                "cost_per_test": 0.02022,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/gpt-5.4-nano-2026-03-17": {
                "accuracy": 77.09,
                "latency": 20.527,
                "stderr": 1.891,
                "cost_per_test": 0.0018,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "alibaba/qwen3.6-plus": {
                "accuracy": 76.963,
                "latency": 173.545,
                "stderr": 1.917,
                "cost_per_test": 0.029294,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "openai/o3-2025-04-16": {
                "accuracy": 76.654,
                "latency": 46.392,
                "stderr": 1.871,
                "cost_per_test": 0.040334,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-3.5-flash": {
                "accuracy": 76.574,
                "latency": 57.075,
                "stderr": 1.923,
                "cost_per_test": 0.166341,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "kimi/kimi-k2.5-thinking": {
                "accuracy": 76.442,
                "latency": 147.155,
                "stderr": 1.986,
                "cost_per_test": 0.02489,
                "temperature": null,
                "top_p": 0.95,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Moonshot AI",
                "harness": null
            },
            "google/gemini-3.1-pro-preview": {
                "accuracy": 76.114,
                "latency": 69.139,
                "stderr": 1.915,
                "cost_per_test": 0.097954,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-sonnet-5": {
                "accuracy": 76.054,
                "latency": 251.703,
                "stderr": 3.05,
                "cost_per_test": 0.433684,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": "max",
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025": {
                "accuracy": 75.824,
                "latency": 4.491,
                "stderr": 1.851,
                "cost_per_test": 0.001332,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "ant/ling-3.0-flash-af-rc3": {
                "accuracy": 75.593,
                "latency": 44.811,
                "stderr": 2.026,
                "cost_per_test": 0.001618,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 131072,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Ant",
                "harness": null
            },
            "deepseek/deepseek-v4-pro": {
                "accuracy": 75.144,
                "latency": 345.723,
                "stderr": 2.002,
                "cost_per_test": 0.053954,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "max",
                "verbosity": null,
                "compute_effort": null,
                "provider": "DeepSeek",
                "harness": null
            },
            "grok/grok-4.3": {
                "accuracy": 74.399,
                "latency": 100.457,
                "stderr": 2.019,
                "cost_per_test": 0.015293,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "anthropic/claude-opus-4-1-20250805-thinking": {
                "accuracy": 73.901,
                "latency": 57.404,
                "stderr": 1.965,
                "cost_per_test": 0.263427,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-2.5-pro": {
                "accuracy": 73.552,
                "latency": 35.909,
                "stderr": 1.91,
                "cost_per_test": 0.046379,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "openai/gpt-5-nano-2025-08-07": {
                "accuracy": 72.865,
                "latency": 112.909,
                "stderr": 1.891,
                "cost_per_test": 0.006961,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "google/gemini-2.5-flash-lite": {
                "accuracy": 72.832,
                "latency": 4.847,
                "stderr": 1.982,
                "cost_per_test": 0.001211,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3-max-2026-01-23": {
                "accuracy": 72.709,
                "latency": 362.378,
                "stderr": 1.905,
                "cost_per_test": 0.085327,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514": {
                "accuracy": 72.411,
                "latency": 25.674,
                "stderr": 1.929,
                "cost_per_test": 0.038973,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "zai/glm-5.1": {
                "accuracy": 72.27,
                "latency": 95.701,
                "stderr": 2.064,
                "cost_per_test": 0.023717,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "xiaomi/mimo-v2.5": {
                "accuracy": 72.151,
                "latency": 20.261,
                "stderr": 1.851,
                "cost_per_test": 0.001351,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Xiaomi",
                "harness": null
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 72.036,
                "latency": 43.386,
                "stderr": 1.9,
                "cost_per_test": 0.061162,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-opus-4-1-20250805": {
                "accuracy": 71.753,
                "latency": 38.04,
                "stderr": 2.021,
                "cost_per_test": 0.187162,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "google/gemini-3.5-flash-lite": {
                "accuracy": 70.886,
                "latency": 18.991,
                "stderr": 2.031,
                "cost_per_test": 0.019867,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "alibaba/qwen3.5-flash": {
                "accuracy": 70.619,
                "latency": 80.406,
                "stderr": 2.09,
                "cost_per_test": 0.004425,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Alibaba",
                "harness": null
            },
            "google/gemini-3-flash-preview": {
                "accuracy": 69.917,
                "latency": 23.491,
                "stderr": 1.899,
                "cost_per_test": 0.014379,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "anthropic/claude-sonnet-4-20250514-thinking": {
                "accuracy": 69.353,
                "latency": 39.572,
                "stderr": 2.212,
                "cost_per_test": 0.053443,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Anthropic",
                "harness": null
            },
            "openai/o4-mini-2025-04-16": {
                "accuracy": 69.139,
                "latency": 81.959,
                "stderr": 1.957,
                "cost_per_test": 0.040605,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "OpenAI",
                "harness": null
            },
            "zai/glm-4.7": {
                "accuracy": 68.629,
                "latency": 167.479,
                "stderr": 2.123,
                "cost_per_test": 0.019082,
                "temperature": 1.0,
                "top_p": 1.0,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Zhipu AI",
                "harness": null
            },
            "mistralai/mistral-medium-3.5": {
                "accuracy": 67.728,
                "latency": 104.053,
                "stderr": 2.011,
                "cost_per_test": 0.157641,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Mistral AI",
                "harness": null
            },
            "google/gemini-2.5-flash-lite-preview-09-2025-thinking": {
                "accuracy": 66.877,
                "latency": 11.518,
                "stderr": 1.923,
                "cost_per_test": 0.002567,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "poolside/laguna-m.1": {
                "accuracy": 65.911,
                "latency": 111.816,
                "stderr": 2.007,
                "cost_per_test": 0.002202,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "google/gemini-3.1-flash-lite-preview": {
                "accuracy": 63.902,
                "latency": 16.793,
                "stderr": 1.823,
                "cost_per_test": 0.002195,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Google",
                "harness": null
            },
            "grok/grok-4.20-0309-reasoning": {
                "accuracy": 63.412,
                "latency": 18.596,
                "stderr": 2.095,
                "cost_per_test": 0.031303,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "SpaceXAI",
                "harness": null
            },
            "poolside/laguna-xs.2": {
                "accuracy": 61.432,
                "latency": 68.824,
                "stderr": 2.349,
                "cost_per_test": 0.001126,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Poolside",
                "harness": null
            },
            "cohere/command-a-plus-05-2026": {
                "accuracy": 55.682,
                "latency": 203.051,
                "stderr": 3.646,
                "cost_per_test": 0.140316,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Cohere",
                "harness": null
            },
            "inception/mercury-2.5": {
                "accuracy": 55.093,
                "latency": 9.992,
                "stderr": 2.095,
                "cost_per_test": 0.004476,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "compute_effort": null,
                "provider": "Inception",
                "harness": null
            },
            "fireworks/llama4-maverick-instruct-basic": {
                "accuracy": 54.219,
                "latency": 25.048,
                "stderr": 1.871,
                "cost_per_test": 0.00246,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            },
            "together/meta-llama/Llama-4-Scout-17B-16E-Instruct": {
                "accuracy": 50.593,
                "latency": 11.323,
                "stderr": 1.901,
                "cost_per_test": 0.0017,
                "temperature": 1.0,
                "top_p": null,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Together AI",
                "harness": null
            },
            "fireworks/nemotron-lightning-3p5-30b-a3b": {
                "accuracy": 4.267,
                "latency": 105.515,
                "stderr": 0.492,
                "cost_per_test": 0.006241,
                "temperature": 1.0,
                "top_p": 0.95,
                "max_output_tokens": 30000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "compute_effort": null,
                "provider": "Fireworks AI",
                "harness": null
            }
        }
    }
}