[
  {
    "model_id": "deepseek-v4-pro",
    "benchmark_id": "hle",
    "variant": "no_tools",
    "value": 41.0,
    "source_url": "https://artificialanalysis.ai/evaluations/humanitys-last-exam",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "AA's own run of 'DeepSeek V4 Pro 0813 (Reasoning, Max Effort)' — text-only, no tools. Replaces the launch chart's self-reported 42.7. AA's run of the older 0424 V4 Pro scored 37.5; don't conflate."
  },
  {
    "model_id": "deepseek-v4-pro",
    "benchmark_id": "hle",
    "variant": "with_tools",
    "value": 60.0,
    "source_url": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813",
    "source_type": "self-reported",
    "date_observed": "2026-08-13",
    "notes": ""
  },
  {
    "model_id": "deepseek-v4-pro",
    "benchmark_id": "terminal-bench-2-1",
    "variant": "default",
    "value": 78.7,
    "source_url": "https://artificialanalysis.ai/evaluations/humanitys-last-exam",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "AA's own harness ('DeepSeek V4 Pro 0813 (max)'). The vendor's launch chart claims 87.9 — 9.2 points above this independent run, and tbench.ai's official board carries no DeepSeek entry at all (board max is 83.8, below the vendor claim)."
  },
  {
    "model_id": "deepseek-v4-pro",
    "benchmark_id": "deepswe",
    "variant": "default",
    "value": 62.7,
    "source_url": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813",
    "source_type": "self-reported",
    "date_observed": "2026-08-13",
    "notes": "Independently corroborated by third-party launch coverage. Up from 12.8 in the April preview build — the steepest single-generation jump on this chart."
  },
  {
    "model_id": "deepseek-v4-pro",
    "benchmark_id": "toolathlon-verified",
    "variant": "default",
    "value": 74.1,
    "source_url": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813",
    "source_type": "self-reported",
    "date_observed": "2026-08-13",
    "notes": ""
  },
  {
    "model_id": "deepseek-v4-pro",
    "benchmark_id": "agents-last-exam",
    "variant": "default",
    "value": 25.2,
    "source_url": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813",
    "source_type": "self-reported",
    "date_observed": "2026-08-17",
    "notes": "Vendor launch-chart claim; the 0813 model is not on Snorkel's board (its only DeepSeek row predates 0813 and scored 12.4 Overall pass rate under OpenClaw). Kept self-reported until an independent 0813 run exists. Note the chart's 25.2 matches no cell on the board."
  },
  {
    "model_id": "claude-opus-4-8",
    "benchmark_id": "hle",
    "variant": "no_tools",
    "value": 48.7,
    "source_url": "https://artificialanalysis.ai/evaluations/humanitys-last-exam",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "AA's own run (Adaptive Reasoning, Max Effort; text-only, no tools). Replaces the self-reported 49.8 from DeepSeek's comparison chart. Distinct from 'Fable 5 (Opus 4.8 fallback)' at 55.5 — a different system."
  },
  {
    "model_id": "claude-opus-4-8",
    "benchmark_id": "hle",
    "variant": "with_tools",
    "value": 57.9,
    "source_url": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813",
    "source_type": "self-reported",
    "date_observed": "2026-08-13",
    "notes": "As published in DeepSeek's comparison chart, not independently verified against Anthropic's own materials."
  },
  {
    "model_id": "claude-opus-4-8",
    "benchmark_id": "terminal-bench-2-1",
    "variant": "default",
    "value": 78.9,
    "source_url": "https://www.tbench.ai/leaderboard/terminal-bench/2.1",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "Official board's verified run ('Claude Code + Opus 4.8', 78.9±1.3, rank 5/17, verified by a Terminal-Bench team member). Vendor chart claimed 85.0. AA's own different-harness run gives 84.6 — harness choice alone spans 6 points here."
  },
  {
    "model_id": "claude-opus-4-8",
    "benchmark_id": "deepswe",
    "variant": "default",
    "value": 59.0,
    "source_url": "https://deepswe.datacurve.ai",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "59%±2 pass@1, mini-swe-agent harness, rank 10 (board updated 2026-08-13). Replaces the self-reported 58.0 — the two agree within error."
  },
  {
    "model_id": "claude-opus-4-8",
    "benchmark_id": "toolathlon-verified",
    "variant": "default",
    "value": 76.2,
    "source_url": "https://toolathlon.xyz/docs/leaderboard",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "Label correction, same value: the board's 76.2±3.4 (rank 2) carries Toolathlon's 'Evaluated by us' badge — this was always an independent run, we had it mislabeled self-reported."
  },
  {
    "model_id": "claude-opus-4-8",
    "benchmark_id": "agents-last-exam",
    "variant": "default",
    "value": 27.0,
    "source_url": "https://snorkel.ai/leaderboard/agents-last-exam/",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "Overall pass rate 27.0 (Claude Code, Max; score 45.1; $3,985). Replaces 25.7 — that value was the ALE-CLI split (and coincidentally the launch chart's claim); normalized to Overall + independent."
  },
  {
    "model_id": "gemini-3-1-pro-preview",
    "benchmark_id": "hle",
    "variant": "no_tools",
    "value": 47.0,
    "source_url": "https://artificialanalysis.ai/evaluations/humanitys-last-exam",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "AA's own run (text-only subset, no tools). Replaces Google's self-reported 44.4 — independent beats vendor at the same setup."
  },
  {
    "model_id": "gemini-3-1-pro-preview",
    "benchmark_id": "hle",
    "variant": "with_tools",
    "value": 51.4,
    "source_url": "https://deepmind.google/models/model-cards/gemini-3-1-pro/",
    "source_type": "self-reported",
    "date_observed": "2026-08-17",
    "notes": "Reported as 'Search + Code tools', a superset of plain tool access."
  },
  {
    "model_id": "gemini-3-1-pro-preview",
    "benchmark_id": "gpqa-diamond",
    "variant": "default",
    "value": 95.45,
    "source_url": "https://www.vals.ai/benchmarks/gpqa",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "vals.ai run, rank 1/133 (updated 2026-08-15). Replaces Google's self-reported 94.3. vals.ai notes 24 models at 90%+ — benchmark near saturation."
  },
  {
    "model_id": "gemini-3-1-pro-preview",
    "benchmark_id": "swe-bench-verified",
    "variant": "default",
    "value": 78.8,
    "source_url": "https://www.vals.ai/benchmarks/swebench",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "vals.ai run, rank 24/83, bash-only mini-swe-agent harness (updated 2026-08-14). Replaces Google's self-reported 80.6."
  },
  {
    "model_id": "gemini-3-1-pro-preview",
    "benchmark_id": "terminal-bench-2-1",
    "variant": "default",
    "value": 65.8,
    "source_url": "https://www.tbench.ai/leaderboard/terminal-bench/2.1",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "Independent leaderboard score (Gemini CLI, 'high' effort). A Terminus-2-harness run on the same leaderboard scores 65.6 — essentially the same. This is the ONLY one of this site's 5 newly-added models with an independent Terminal-Bench 2.1 submission; the other four only have unverified vendor-reported numbers 15-23 points higher."
  },
  {
    "model_id": "gemini-3-1-pro-preview",
    "benchmark_id": "toolathlon-verified",
    "variant": "default",
    "value": 61.1,
    "source_url": "https://toolathlon.xyz/docs/leaderboard",
    "source_type": "independent",
    "date_observed": "2026-07-01",
    "notes": ""
  },
  {
    "model_id": "gemini-3-1-pro-preview",
    "benchmark_id": "agents-last-exam",
    "variant": "default",
    "value": 16.4,
    "source_url": "https://snorkel.ai/leaderboard/agents-last-exam/",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "Overall pass rate 16.4 (Gemini CLI, High; score 32.7; $2,018). Corrects our earlier 32.7 — that was score_pct, not pass rate."
  },
  {
    "model_id": "kimi-k3",
    "benchmark_id": "hle",
    "variant": "no_tools",
    "value": 46.9,
    "source_url": "https://artificialanalysis.ai/evaluations/humanitys-last-exam",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "AA's own run, 'Kimi K3 (max)' (text-only, no tools). Replaces Moonshot's self-reported 43.5."
  },
  {
    "model_id": "kimi-k3",
    "benchmark_id": "hle",
    "variant": "with_tools",
    "value": 56.0,
    "source_url": "https://huggingface.co/moonshotai/Kimi-K3/blob/main/README.md",
    "source_type": "self-reported",
    "date_observed": "2026-08-17",
    "notes": ""
  },
  {
    "model_id": "kimi-k3",
    "benchmark_id": "gpqa-diamond",
    "variant": "default",
    "value": 92.93,
    "source_url": "https://www.vals.ai/benchmarks/gpqa",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "vals.ai run, rank 10/133 (updated 2026-08-15). Replaces Moonshot's self-reported 93.5."
  },
  {
    "model_id": "kimi-k3",
    "benchmark_id": "terminal-bench-2-1",
    "variant": "default",
    "value": 88.3,
    "source_url": "https://github.com/MoonshotAI/Kimi-K3/blob/main/README.md",
    "source_type": "self-reported",
    "date_observed": "2026-08-17",
    "notes": "Vendor-reported (harness: Kimi Code). Does not appear on the official independent Terminal-Bench 2.1 leaderboard as of Aug 2026 — treat as unverified relative to Gemini 3.1 Pro's 65.8 independent score on the same benchmark."
  },
  {
    "model_id": "kimi-k3",
    "benchmark_id": "deepswe",
    "variant": "default",
    "value": 69,
    "source_url": "https://deepswe.datacurve.ai/",
    "source_type": "independent",
    "date_observed": "2026-08-13",
    "notes": "Independent score (69% ± 5%, $4.65/task); vendor self-reported 67.5 is close and consistent."
  },
  {
    "model_id": "kimi-k3",
    "benchmark_id": "toolathlon-verified",
    "variant": "default",
    "value": 76.5,
    "source_url": "https://toolathlon.xyz/docs/leaderboard",
    "source_type": "independent",
    "date_observed": "2026-07-16",
    "notes": "Rank 1 on the independent leaderboard; exact match with vendor self-reported figure."
  },
  {
    "model_id": "kimi-k3",
    "benchmark_id": "agents-last-exam",
    "variant": "default",
    "value": 28.3,
    "source_url": "https://snorkel.ai/leaderboard/agents-last-exam/",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "Overall pass rate 28.3, board rank 3 (Kimi Code, Max; score 51.6; $606). Metric verified correct in the 2026-08-17 normalization."
  },
  {
    "model_id": "qwen3-8-max",
    "benchmark_id": "hle",
    "variant": "no_tools",
    "value": 43.0,
    "source_url": "https://artificialanalysis.ai/evaluations/humanitys-last-exam",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "AA's own run, 'Qwen3.8 Max' (text-only, no tools). Replaces Alibaba's self-reported 43.6 — the two agree closely here."
  },
  {
    "model_id": "qwen3-8-max",
    "benchmark_id": "hle",
    "variant": "with_tools",
    "value": 56.2,
    "source_url": "https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/main/README.md",
    "source_type": "self-reported",
    "date_observed": "2026-08-17",
    "notes": ""
  },
  {
    "model_id": "qwen3-8-max",
    "benchmark_id": "gpqa-diamond",
    "variant": "default",
    "value": 93.69,
    "source_url": "https://www.vals.ai/benchmarks/gpqa",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "vals.ai run, rank 5/133 (updated 2026-08-15). Replaces Alibaba's self-reported 92.6."
  },
  {
    "model_id": "qwen3-8-max",
    "benchmark_id": "terminal-bench-2-1",
    "variant": "default",
    "value": 86.6,
    "source_url": "https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/main/README.md",
    "source_type": "self-reported",
    "date_observed": "2026-08-17",
    "notes": "Vendor-reported (harness: Claude Code, avg@10). Does not appear on the official independent Terminal-Bench 2.1 leaderboard as of Aug 2026."
  },
  {
    "model_id": "qwen3-8-max",
    "benchmark_id": "deepswe",
    "variant": "default",
    "value": 57,
    "source_url": "https://deepswe.datacurve.ai/",
    "source_type": "independent",
    "date_observed": "2026-08-13",
    "notes": "Independent score (57% ± 3%, $3.73/task) — debuted as the highest-scoring new entry on this leaderboard at the time."
  },
  {
    "model_id": "qwen3-8-max",
    "benchmark_id": "toolathlon-verified",
    "variant": "default",
    "value": 72.5,
    "source_url": "https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/main/README.md",
    "source_type": "self-reported",
    "date_observed": "2026-08-17",
    "notes": "Vendor-reported. Does not appear on the official Toolathlon-Verified leaderboard as of Aug 2026."
  },
  {
    "model_id": "qwen3-8-max",
    "benchmark_id": "agents-last-exam",
    "variant": "default",
    "value": 27.0,
    "source_url": "https://snorkel.ai/leaderboard/agents-last-exam/",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "Overall pass rate 27.0 (Claude Code, XHigh; score 52.5; $486). Corrects our earlier 52.5 — that was score_pct, not pass rate."
  },
  {
    "model_id": "glm-5-2",
    "benchmark_id": "hle",
    "variant": "no_tools",
    "value": 41.1,
    "source_url": "https://artificialanalysis.ai/evaluations/humanitys-last-exam",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "AA's own run, 'GLM-5.2 (max)' (text-only, no tools). Replaces Z.ai's self-reported 40.5."
  },
  {
    "model_id": "glm-5-2",
    "benchmark_id": "hle",
    "variant": "with_tools",
    "value": 54.7,
    "source_url": "https://huggingface.co/zai-org/GLM-5.2/blob/main/README.md",
    "source_type": "self-reported",
    "date_observed": "2026-08-17",
    "notes": "Corroborated by https://docs.z.ai/guides/llm/glm-5.2."
  },
  {
    "model_id": "glm-5-2",
    "benchmark_id": "gpqa-diamond",
    "variant": "default",
    "value": 85.61,
    "source_url": "https://www.vals.ai/benchmarks/gpqa",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "vals.ai run, rank 42/133 (updated 2026-08-15). Replaces Z.ai's self-reported 91.2 — a 5.6-point gap between vendor claim and independent run, the largest such gap on this site."
  },
  {
    "model_id": "glm-5-2",
    "benchmark_id": "terminal-bench-2-1",
    "variant": "default",
    "value": 81.0,
    "source_url": "https://huggingface.co/zai-org/GLM-5.2/blob/main/README.md",
    "source_type": "self-reported",
    "date_observed": "2026-08-17",
    "notes": "Vendor-reported (Terminus-2 harness). Only the prior GLM-5.1 (58.7) appears on the official independent Terminal-Bench 2.1 leaderboard as of Aug 2026 — GLM-5.2 itself is unverified there."
  },
  {
    "model_id": "glm-5-2",
    "benchmark_id": "deepswe",
    "variant": "default",
    "value": 44,
    "source_url": "https://deepswe.datacurve.ai/",
    "source_type": "independent",
    "date_observed": "2026-08-13",
    "notes": "Independent score (44% ± 2%, $3.92/task)."
  },
  {
    "model_id": "glm-5-2",
    "benchmark_id": "toolathlon-verified",
    "variant": "default",
    "value": 59.9,
    "source_url": "https://toolathlon.xyz/docs/leaderboard",
    "source_type": "independent",
    "date_observed": "2026-06-30",
    "notes": ""
  },
  {
    "model_id": "glm-5-2",
    "benchmark_id": "agents-last-exam",
    "variant": "default",
    "value": 20.4,
    "source_url": "https://snorkel.ai/leaderboard/agents-last-exam/",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "Overall pass rate 20.4 (Claude Code, Max; score 41.1; $1,086). Corrects our earlier 41.1 — that was score_pct, not pass rate."
  },
  {
    "model_id": "gpt-5-6-sol",
    "benchmark_id": "deepswe",
    "variant": "default",
    "value": 73,
    "source_url": "https://deepswe.datacurve.ai/",
    "source_type": "independent",
    "date_observed": "2026-08-13",
    "notes": "Independent score (73% ± 3%, $8.39/task, max reasoning effort), rank #2 of 17 tracked models. Found via the independent leaderboard directly, bypassing openai.com (which blocks automated access) entirely."
  },
  {
    "model_id": "gpt-5-6-sol",
    "benchmark_id": "agents-last-exam",
    "variant": "default",
    "value": 30.6,
    "source_url": "https://snorkel.ai/leaderboard/agents-last-exam/",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "Overall pass rate 30.6, board rank 1 (Codex, XHigh; score 53.6; $772). Corrects our earlier 53.6, which was the score_pct (partial-credit) metric, not the pass rate — metric mix-up, not a model change."
  },
  {
    "model_id": "claude-opus-5",
    "benchmark_id": "hle",
    "variant": "no_tools",
    "value": 54.9,
    "source_url": "https://artificialanalysis.ai/articles/opus-5",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "AA live leaderboard value (max effort, text-only, no tools), board-verified 2026-08-17; AA's launch article had reported 53.0."
  },
  {
    "model_id": "claude-opus-5",
    "benchmark_id": "deepswe",
    "variant": "default",
    "value": 74.0,
    "source_url": "https://deepswe.datacurve.ai",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "74%±4 pass rate on the mini-swe-agent harness (all models run identically); board updated 2026-08-13."
  },
  {
    "model_id": "claude-opus-5",
    "benchmark_id": "swe-bench-verified",
    "variant": "default",
    "value": 97.0,
    "source_url": "https://www.vals.ai/benchmarks/swebench",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "vals.ai run, rank 1 of 83 (updated 2026-08-14); Anthropic self-reports 96.0. Benchmark near-saturated — five models at 95%+."
  },
  {
    "model_id": "claude-opus-5",
    "benchmark_id": "livecodebench",
    "variant": "default",
    "value": 89.03,
    "source_url": "https://www.vals.ai/benchmarks/lcb",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "vals.ai, updated 2026-08-15; re-verified on the vals.ai model page 2026-08-17."
  },
  {
    "model_id": "claude-fable-5",
    "benchmark_id": "hle",
    "variant": "no_tools",
    "value": 55.5,
    "source_url": "https://artificialanalysis.ai/articles/opus-5",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "AA live leaderboard value (Adaptive Reasoning, Max Effort, Opus 4.8 Fallback — the shipped system: classifier-refused tasks fall back to Opus 4.8, ~9%, included in the score). Board-verified 2026-08-17; AA's launch article had reported 53.0."
  },
  {
    "model_id": "claude-fable-5",
    "benchmark_id": "terminal-bench-2-1",
    "variant": "default",
    "value": 83.8,
    "source_url": "https://www.tbench.ai/leaderboard/terminal-bench/2.1",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "Rank 1 of 17 (83.8%±1.2, Claude Code agent, submitted 2026-06-07 pre-GA). Anthropic's own launch table claims 88.0."
  },
  {
    "model_id": "claude-fable-5",
    "benchmark_id": "deepswe",
    "variant": "default",
    "value": 70.0,
    "source_url": "https://deepswe.datacurve.ai",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "70%±4 pass rate, mini-swe-agent harness; board updated 2026-08-13. Costs $21.63/task vs Opus 5's $11.84."
  },
  {
    "model_id": "claude-fable-5",
    "benchmark_id": "gpqa-diamond",
    "variant": "default",
    "value": 93.18,
    "source_url": "https://www.vals.ai/benchmarks/gpqa",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "Major caveat, per vals.ai itself: 93.18% counts refusal-triggered fallbacks as successes — counting refusals as failures drops it to 55.56%. Read with the benchmark's saturation grade in mind."
  },
  {
    "model_id": "claude-fable-5",
    "benchmark_id": "agents-last-exam",
    "variant": "default",
    "value": 25.7,
    "source_url": "https://snorkel.ai/leaderboard/agents-last-exam/",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "Overall pass rate 25.7 (Claude Code, XHigh; partial-credit score 48.7; $4,340 eval cost). Snorkel flags the served variant may understate the full capability tier. Board as_of 2026-08-14."
  },
  {
    "model_id": "claude-fable-5",
    "benchmark_id": "livecodebench",
    "variant": "default",
    "value": 89.78,
    "source_url": "https://www.vals.ai/benchmarks/lcb",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "vals.ai top performer, updated 2026-08-15."
  },
  {
    "model_id": "claude-sonnet-5",
    "benchmark_id": "hle",
    "variant": "no_tools",
    "value": 41.3,
    "source_url": "https://artificialanalysis.ai/evaluations/humanitys-last-exam?models=claude-sonnet-5",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "Artificial Analysis's own run (max effort, text-only, no tools; 2,158 questions, pass@1). Not comparable to vendors' with-tools HLE claims."
  },
  {
    "model_id": "claude-sonnet-5",
    "benchmark_id": "terminal-bench-2-1",
    "variant": "default",
    "value": 74.6,
    "source_url": "https://www.tbench.ai/leaderboard/terminal-bench/2.1",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "74.6%±1.6, rank 10 (Claude Code agent, 2026-07-09). Cross-corroborated by vals.ai's independent run at 74.53%."
  },
  {
    "model_id": "claude-sonnet-5",
    "benchmark_id": "deepswe",
    "variant": "default",
    "value": 54.0,
    "source_url": "https://deepswe.datacurve.ai",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "54%±4 pass@1 on the mini-swe-agent harness, rank 13/17; $26.40 avg cost/task. Board updated 2026-08-13."
  },
  {
    "model_id": "claude-sonnet-5",
    "benchmark_id": "swe-bench-verified",
    "variant": "default",
    "value": 79.6,
    "source_url": "https://www.vals.ai/benchmarks/swebench",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "79.60%±1.80, rank 22/83 (updated 2026-08-14). vals.ai's earlier Jun-30 setup gave 75.49 — harness changed since, numbers not directly comparable across dates."
  },
  {
    "model_id": "claude-sonnet-5",
    "benchmark_id": "gpqa-diamond",
    "variant": "default",
    "value": 88.89,
    "source_url": "https://www.vals.ai/benchmarks/gpqa",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "88.89%±2.22, rank 30/133 (updated 2026-08-15). vals.ai notes the benchmark is largely saturated — 24 models at 90%+."
  },
  {
    "model_id": "claude-sonnet-5",
    "benchmark_id": "livecodebench",
    "variant": "default",
    "value": 82.43,
    "source_url": "https://www.vals.ai/benchmarks/lcb",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "82.43%±1.09 on v6, rank 50/138 (updated 2026-08-15)."
  },
  {
    "model_id": "gpt-5-6-sol",
    "benchmark_id": "hle",
    "variant": "no_tools",
    "value": 49.5,
    "source_url": "https://artificialanalysis.ai/evaluations/humanitys-last-exam",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "AA's own run, 'GPT-5.6 Sol (max)' (text-only, no tools). Effort matters a lot on this model: AA also lists high 46.0, medium 42.2, low 39.4."
  },
  {
    "model_id": "gpt-5-6-sol",
    "benchmark_id": "swe-bench-verified",
    "variant": "default",
    "value": 96.2,
    "source_url": "https://www.vals.ai/benchmarks/swebench",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "vals.ai run, rank 3/83, bash-only harness, $1.15/test (updated 2026-08-14)."
  },
  {
    "model_id": "gpt-5-6-sol",
    "benchmark_id": "gpqa-diamond",
    "variant": "default",
    "value": 95.2,
    "source_url": "https://www.vals.ai/benchmarks/gpqa",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "vals.ai run, rank 2/133 (updated 2026-08-15)."
  },
  {
    "model_id": "gpt-5-6-sol",
    "benchmark_id": "livecodebench",
    "variant": "default",
    "value": 82.6,
    "source_url": "https://www.vals.ai/benchmarks/lcb",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "vals.ai run, rank 49/138 — notably low vs its SWE/GPQA ranks; the 56.5s avg latency suggests provider-default rather than max reasoning config (updated 2026-08-15)."
  },
  {
    "model_id": "kimi-k3",
    "benchmark_id": "swe-bench-verified",
    "variant": "default",
    "value": 93.4,
    "source_url": "https://www.vals.ai/benchmarks/swebench",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "vals.ai run, rank 6/83, bash-only harness, $0.76/test (updated 2026-08-14)."
  },
  {
    "model_id": "qwen3-8-max",
    "benchmark_id": "swe-bench-verified",
    "variant": "default",
    "value": 85.6,
    "source_url": "https://www.vals.ai/benchmarks/swebench",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "vals.ai run, rank 13/83, bash-only harness; slowest of the top group at 41m42s/test (updated 2026-08-14)."
  },
  {
    "model_id": "glm-5-2",
    "benchmark_id": "swe-bench-verified",
    "variant": "default",
    "value": 82.8,
    "source_url": "https://www.vals.ai/benchmarks/swebench",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "vals.ai run, rank 14/83, bash-only harness, $0.71/test (updated 2026-08-14)."
  },
  {
    "model_id": "gemini-3-1-pro-preview",
    "benchmark_id": "livecodebench",
    "variant": "default",
    "value": 88.48,
    "source_url": "https://www.vals.ai/benchmarks/lcb",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "vals.ai run, rank 4/138 (updated 2026-08-15)."
  },
  {
    "model_id": "qwen3-8-max",
    "benchmark_id": "livecodebench",
    "variant": "default",
    "value": 87.85,
    "source_url": "https://www.vals.ai/benchmarks/lcb",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "vals.ai run, rank 8/138 (updated 2026-08-15)."
  },
  {
    "model_id": "kimi-k3",
    "benchmark_id": "livecodebench",
    "variant": "default",
    "value": 87.19,
    "source_url": "https://www.vals.ai/benchmarks/lcb",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "vals.ai run, rank 16/138 (updated 2026-08-15)."
  },
  {
    "model_id": "glm-5-2",
    "benchmark_id": "livecodebench",
    "variant": "default",
    "value": 69.5,
    "source_url": "https://www.vals.ai/benchmarks/lcb",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "vals.ai run, rank 88/138 (updated 2026-08-15)."
  },
  {
    "model_id": "deepseek-v4-pro",
    "benchmark_id": "swe-bench-verified",
    "variant": "default",
    "value": 96.4,
    "source_url": "https://www.vals.ai/benchmarks/swebench",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "vals.ai run of the exact 0813 model, rank 2/83 (482/500 resolved), bash-only harness (updated 2026-08-14)."
  },
  {
    "model_id": "claude-opus-4-8",
    "benchmark_id": "swe-bench-verified",
    "variant": "default",
    "value": 88.6,
    "source_url": "https://www.vals.ai/benchmarks/swebench",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "vals.ai run, rank 9/83 (updated 2026-08-14). The board lists a second cheaper-config Opus 4.8 row at 85.8; 88.6 is the primary entry."
  },
  {
    "model_id": "deepseek-v4-pro",
    "benchmark_id": "gpqa-diamond",
    "variant": "default",
    "value": 92.42,
    "source_url": "https://www.vals.ai/benchmarks/gpqa",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "vals.ai run of the 0813 model, rank 15/133 (updated 2026-08-15) — displayed tied with Claude Opus 4.8."
  },
  {
    "model_id": "claude-opus-4-8",
    "benchmark_id": "gpqa-diamond",
    "variant": "default",
    "value": 92.42,
    "source_url": "https://www.vals.ai/benchmarks/gpqa",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "vals.ai run, rank 14/133, displayed tied with DeepSeek V4 Pro 0813 (updated 2026-08-15)."
  },
  {
    "model_id": "deepseek-v4-pro",
    "benchmark_id": "livecodebench",
    "variant": "default",
    "value": 87.53,
    "source_url": "https://www.vals.ai/benchmarks/lcb",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "vals.ai run of the 0813 model, rank 11/138 (updated 2026-08-15). Distinct from base 'DeepSeek V4' at 87.48."
  },
  {
    "model_id": "claude-opus-4-8",
    "benchmark_id": "livecodebench",
    "variant": "default",
    "value": 87.82,
    "source_url": "https://www.vals.ai/benchmarks/lcb",
    "source_type": "independent",
    "date_observed": "2026-08-17",
    "notes": "vals.ai run, rank 9/138 (updated 2026-08-15)."
  }
]