{
  "id": 4422,
  "url": "https://arxiv.org/abs/2605.18840v2",
  "title": "The Growing Pains of Frontier Models: When Leaderboards Stop Separating and What to Measure Next",
  "summary": "Leaderboards rank frontier models on independent axes but do not reveal whether capabilities reinforce or trade off across releases -- and at the frontier, this interaction is the more informative signal. We decompose paired SWE-bench and GPQA Diamond scores into a population coupling trend and per-release residual ($h$-field) that diagnoses capability emphasis from two public benchmark scores. Across 34 models from 10 labs (2024--2026), capabilities cooperate ($r = +0.72$, $p < 10^{-6}$), but c",
  "authors": "Adil Amin",
  "category": "research",
  "topics": "healthcare",
  "orgs": null,
  "regions": null,
  "published_at": "2026-05-13T03:19:38.000Z",
  "fetched_at": "2026-07-14T16:30:59.237Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/4422",
  "original_url": "https://arxiv.org/abs/2605.18840v2",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}