{
    "metadata": {
        "benchmark": "Vibe Code Bench v1",
        "slug": "vibe-code-v1",
        "description": "Can models build web applications from scratch?",
        "benchmark_id": "vibe_code_bench_v1",
        "updated": "2026-02-20",
        "dataset_type": "private",
        "industry": "coding",
        "mode": "agentic",
        "tasks": {
            "overall": "Overall"
        },
        "total_models": 17,
        "models": [
            "alibaba/qwen3-max",
            "anthropic/claude-haiku-4-5-20251001-thinking",
            "anthropic/claude-opus-4-5-20251101-thinking",
            "anthropic/claude-opus-4-6-thinking",
            "anthropic/claude-sonnet-4-5-20250929-thinking",
            "anthropic/claude-sonnet-4-6",
            "google/gemini-2.5-pro",
            "google/gemini-3-pro-preview",
            "grok/grok-4-1-fast-reasoning",
            "grok/grok-4-fast-reasoning",
            "openai/gpt-5-2025-08-07",
            "openai/gpt-5-mini-2025-08-07",
            "openai/gpt-5.1-2025-11-13",
            "openai/gpt-5.1-codex",
            "openai/gpt-5.1-codex-max",
            "openai/gpt-5.2-2025-12-11",
            "zai/glm-4.6"
        ],
        "partners": [],
        "showBadge": false,
        "visible": true,
        "use_cost_per_test": true,
        "family": "vibe_code_bench",
        "version": "1",
        "archived": true
    },
    "tasks": {
        "overall": {
            "openai/gpt-5.2-2025-12-11": {
                "accuracy": 41.3062,
                "latency": 21269.3085,
                "stderr": 0,
                "cost_per_test": 27.13156,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "xhigh",
                "verbosity": null,
                "provider": "OpenAI"
            },
            "anthropic/claude-opus-4-6-thinking": {
                "accuracy": 36.119,
                "latency": 3319.9079,
                "stderr": 0,
                "cost_per_test": 11.008924,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "provider": "Anthropic"
            },
            "anthropic/claude-sonnet-4-6": {
                "accuracy": 29.058,
                "latency": 4138.4229,
                "stderr": 0,
                "cost_per_test": 11.602877,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": true,
                "reasoning_effort": null,
                "verbosity": null,
                "provider": "Anthropic"
            },
            "openai/gpt-5.1-2025-11-13": {
                "accuracy": 24.6056,
                "latency": 1836.1882,
                "stderr": 0,
                "cost_per_test": 2.574926,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "provider": "OpenAI"
            },
            "anthropic/claude-sonnet-4-5-20250929-thinking": {
                "accuracy": 22.6213,
                "latency": 2962.1333,
                "stderr": 0,
                "cost_per_test": 6.656984,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "provider": "Anthropic"
            },
            "openai/gpt-5.1-codex-max": {
                "accuracy": 22.1679,
                "latency": 31182.9378,
                "stderr": 0,
                "cost_per_test": 7.324105,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "provider": "OpenAI"
            },
            "anthropic/claude-opus-4-5-20251101-thinking": {
                "accuracy": 20.6303,
                "latency": 2283.1417,
                "stderr": 0,
                "cost_per_test": 32.873283,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "provider": "Anthropic"
            },
            "openai/gpt-5-2025-08-07": {
                "accuracy": 20.0882,
                "latency": 1852.1489,
                "stderr": 0,
                "cost_per_test": 1.525167,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "provider": "OpenAI"
            },
            "google/gemini-3-pro-preview": {
                "accuracy": 14.2999,
                "latency": 10398.5452,
                "stderr": 0,
                "cost_per_test": 6.417592,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "provider": "Google"
            },
            "openai/gpt-5.1-codex": {
                "accuracy": 13.1152,
                "latency": 3026.0695,
                "stderr": 0,
                "cost_per_test": 3.797914,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "provider": "OpenAI"
            },
            "anthropic/claude-haiku-4-5-20251001-thinking": {
                "accuracy": 11.6784,
                "latency": 1912.9196,
                "stderr": 0,
                "cost_per_test": 1.600392,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 64000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "provider": "Anthropic"
            },
            "openai/gpt-5-mini-2025-08-07": {
                "accuracy": 4.868,
                "latency": 2605.9069,
                "stderr": 0,
                "cost_per_test": 0.176563,
                "temperature": null,
                "top_p": null,
                "max_output_tokens": 128000,
                "reasoning": null,
                "reasoning_effort": "high",
                "verbosity": null,
                "provider": "OpenAI"
            },
            "alibaba/qwen3-max": {
                "accuracy": 3.5058,
                "latency": 3465.1704,
                "stderr": 0,
                "cost_per_test": 6.023105,
                "temperature": 0.7,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "provider": "Alibaba"
            },
            "zai/glm-4.6": {
                "accuracy": 3.0905,
                "latency": 10002.4456,
                "stderr": 0,
                "cost_per_test": 10.845475,
                "temperature": 0.6,
                "top_p": 1,
                "max_output_tokens": 122880,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "provider": "Zhipu AI"
            },
            "google/gemini-2.5-pro": {
                "accuracy": 0.4,
                "latency": 2097.1444,
                "stderr": 0,
                "cost_per_test": 1.223831,
                "temperature": 1,
                "top_p": null,
                "max_output_tokens": 65536,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "provider": "Google"
            },
            "grok/grok-4-1-fast-reasoning": {
                "accuracy": 0,
                "latency": 554.5718,
                "stderr": 0,
                "cost_per_test": 0.120503,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "provider": "SpaceXAI"
            },
            "grok/grok-4-fast-reasoning": {
                "accuracy": 0,
                "latency": 1448.8552,
                "stderr": 0,
                "cost_per_test": 0.133206,
                "temperature": 0.7,
                "top_p": 0.95,
                "max_output_tokens": 2000000,
                "reasoning": null,
                "reasoning_effort": null,
                "verbosity": null,
                "provider": "SpaceXAI"
            }
        }
    }
}
