{
  "id": 6772,
  "url": "https://arxiv.org/abs/2603.24124v2",
  "title": "The Alignment Tax: Response Homogenization in Aligned LLMs and Its Implications for Uncertainty Estimation",
  "summary": "RLHF-aligned language models exhibit response homogenization: on TruthfulQA (n=790), 40-79% of questions produce a single semantic cluster across 10 i.i.d. samples. On affected questions, sampling-based uncertainty methods have zero discriminative power (AUROC=0.500), while free token entropy retains signal (0.603). This alignment tax is task-dependent: on GSM8K (n=500), token entropy achieves 0.724 (Cohen's d=0.81). A base-vs-instruct ablation confirms the causal role of alignment: the base mod",
  "authors": "Mingyi Liu",
  "category": "research",
  "topics": "bias-fairness,safety-alignment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-03-25T09:35:15.000Z",
  "fetched_at": "2026-07-14T16:32:45.891Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/6772",
  "original_url": "https://arxiv.org/abs/2603.24124v2",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}