[
  {
    "id": "claude-opus-4-8-vs-deepseek-v4-pro",
    "model_a": "claude-opus-4-8",
    "model_b": "deepseek-v4-pro",
    "verdict": "Independent runs have rewritten this matchup, mostly into ties. Claude Opus 4.8's one clean, confirmed win is raw reasoning without tools (HLE 48.7 vs 41.0, both Artificial Analysis runs). DeepSeek V4 Pro posts a bigger SWE-bench Verified number (96.4 vs 88.6, both vals.ai) — but that benchmark is saturated (grade D here), and by our own methodology rankings on it are unreliable regardless of gap size. Everything else — terminal work, long-horizon coding, multi-tool chores, professional tasks, GPQA, LiveCodeBench — lands inside the noise band. The launch chart's drama mostly dissolves under independent measurement; what survives is a price gap: DeepSeek is roughly 7x cheaper per output token at parity.",
    "who_picks": [
      {
        "who": "High-volume or budget-constrained workloads",
        "pick": "deepseek-v4-pro",
        "why": "measured parity at about 7x cheaper output tokens — the ties are the story: you rarely give up confirmed capability for the discount"
      },
      {
        "who": "Single-shot reasoning without tool access",
        "pick": "claude-opus-4-8",
        "why": "the matchup's only clean independently-confirmed real gap: HLE no-tools 48.7 vs 41.0 (Artificial Analysis ran both)"
      },
      {
        "who": "Repository-scale bug fixing",
        "pick": "deepseek-v4-pro",
        "why": "96.4 vs 88.6 on vals.ai's identical harness — but flagged honestly: SWE-bench Verified is saturated (grade D), so treat this as weak evidence, and the price still favors DeepSeek anyway"
      }
    ],
    "verdict_basis": "d030248b5be5"
  },
  {
    "id": "claude-opus-4-8-vs-gemini-3-1-pro-preview",
    "model_a": "claude-opus-4-8",
    "model_b": "gemini-3-1-pro-preview",
    "verdict": "Independent runs now give Claude Opus 4.8 three clean real wins on agent work, each confirmed on both sides: Terminal-Bench 2.1 (78.9 vs 65.8), Toolathlon-Verified (76.2 vs 61.1), and Agents' Last Exam (27.0 vs 16.4 overall pass rate). The reasoning and coding picture is all statistical ties — HLE without tools, GPQA Diamond, LiveCodeBench. (An earlier version of this page said Agents' Last Exam favored Gemini; that was our error — we had mixed the board's partial-credit score with its pass rate. Corrected 2026-08-17.) Gemini's remaining case is price: $2/$12 vs $5/$25.",
    "who_picks": [
      {
        "who": "Agentic work — terminal, multi-tool, professional tasks",
        "pick": "claude-opus-4-8",
        "why": "three real gaps, every number on both sides independently run (tbench.ai, toolathlon.xyz, Snorkel)"
      },
      {
        "who": "Budget-sensitive reasoning and coding",
        "pick": "gemini-3-1-pro-preview",
        "why": "ties Opus 4.8 on HLE no-tools, GPQA and LiveCodeBench at less than half the price"
      },
      {
        "who": "Want every number verifiable",
        "pick": "claude-opus-4-8",
        "why": "since independent boards covered both models, Opus 4.8's wins are the ones that survived — the earlier 'only Gemini is verified' framing is obsolete"
      }
    ],
    "verdict_basis": "309d8566a09e"
  },
  {
    "id": "deepseek-v4-pro-vs-glm-5-2",
    "model_a": "deepseek-v4-pro",
    "model_b": "glm-5-2",
    "verdict": "DeepSeek V4 Pro leads on most of what independent boards can check: LiveCodeBench (87.53 vs 69.5, both vals.ai runs — a real gap), long-horizon coding and multi-tool chores (real-sized gaps, though DeepSeek's side of those two is still the vendor's number), and Agents' Last Exam (25.2 vs an independently-measured 20.4 — an earlier version of this page had GLM ahead here; that was our metric mix-up, corrected 2026-08-17). Reasoning is a dead heat: HLE without tools is 41.0 vs 41.1 in Artificial Analysis's own runs. DeepSeek is also the cheaper model. GLM-5.2's real differentiator isn't a benchmark: it's MIT-licensed open weights.",
    "who_picks": [
      {
        "who": "Most workloads on a budget",
        "pick": "deepseek-v4-pro",
        "why": "leads or ties everywhere measured, and its listed rates are cheaper than GLM-5.2's"
      },
      {
        "who": "Self-hosting, fine-tuning, or license-sensitive deployment",
        "pick": "glm-5-2",
        "why": "MIT open weights — the one axis where no benchmark matters and GLM wins outright"
      },
      {
        "who": "Contest-style coding",
        "pick": "deepseek-v4-pro",
        "why": "LiveCodeBench 87.53 vs 69.5, both sides run by vals.ai — the cleanest real gap in this matchup"
      }
    ],
    "verdict_basis": "9bd191fd2a2a"
  },
  {
    "id": "gemini-3-1-pro-preview-vs-kimi-k3",
    "model_a": "gemini-3-1-pro-preview",
    "model_b": "kimi-k3",
    "verdict": "Independent runs now hand Kimi K3 the agent-work sweep: Toolathlon-Verified 76.5 vs 61.1 and Agents' Last Exam 28.3 vs 16.4 overall pass rate — both real gaps, every number independently run. (An earlier version of this page said Gemini led Agents' Last Exam; that was our metric mix-up — partial-credit score vs pass rate — corrected 2026-08-17.) On Terminal-Bench the gap looks huge (88.3 vs 65.8) but Kimi's side is still its own claim, so keep skepticism there. Gemini 3.1 Pro ties on the reasoning/coding trio — HLE no-tools, GPQA, LiveCodeBench — and costs less ($2/$12 vs $3/$15).",
    "who_picks": [
      {
        "who": "Multi-tool chores and professional agent tasks",
        "pick": "kimi-k3",
        "why": "two real gaps with both sides independently run (toolathlon.xyz, Snorkel) — the confirmed part of the sweep"
      },
      {
        "who": "Reasoning and contest coding on a budget",
        "pick": "gemini-3-1-pro-preview",
        "why": "statistical ties on HLE no-tools, GPQA and LiveCodeBench at a lower list price"
      },
      {
        "who": "Terminal-heavy work",
        "pick": "kimi-k3",
        "why": "hedged pick: the 22-point lead is real-sized but Kimi's number is self-reported — if that risk bothers you, the verified runner-up is Opus 4.8's 78.9, not Gemini"
      }
    ],
    "verdict_basis": "dd33e210aa3c"
  }
]