{
  "id": 5287,
  "url": "https://arxiv.org/abs/2604.25077v1",
  "title": "Evaluating Risks in Weak-to-Strong Alignment: A Bias-Variance Perspective",
  "summary": "Weak-to-strong alignment offers a promising route to scalable supervision, but it can fail when a strong model becomes confidently wrong on examples that lie in the weak teacher's blind spots. Understanding such failures requires going beyond aggregate accuracy, since weak-to-strong errors depend not only on whether the strong model disagrees with its teacher, but also on how confidence and uncertainty are distributed across examples. In this work, we analyze weak-to-strong alignment through a b",
  "authors": "Hamid Osooli, Kareema Batool, Rick Gentry, Tiasa Singha Roy, Ashwin Gupta, Anirudha Ramesh",
  "category": "research",
  "topics": "bias-fairness,safety-alignment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-04-28T00:15:23.000Z",
  "fetched_at": "2026-07-14T16:31:40.219Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/5287",
  "original_url": "https://arxiv.org/abs/2604.25077v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}