{
  "id": 17937,
  "url": "https://arxiv.org/abs/2608.07419v1",
  "title": "Beyond Post-Hoc Temperature Scaling: Bilevel Optimization for LLM Calibration",
  "summary": "Preference alignment often makes large language models (LLMs) overconfident and poorly calibrated. Traditional post-hoc temperature scaling is inherently domain-dependent: a temperature fitted on one domain does not generalize across domains. This motivates us to modify model parameters during training to improve calibration. We propose maximizing the entropy of predictive distributions as the calibration objective, which directly targets overconfidence by discouraging overly concentrated predic",
  "authors": "Ruochen Jin, Zhanliang Wang, Zongyu Dai, Jiancong Xiao, Bojian Hou",
  "category": "research",
  "topics": "safety-alignment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-07T17:05:10.000Z",
  "fetched_at": "2026-08-10T05:10:00.488Z",
  "source_slug": "arxiv-cslg",
  "source_name": "arXiv cs.LG",
  "source_homepage": "https://arxiv.org/list/cs.LG/recent",
  "ethics_ai_record_url": "https://ethics.ai/record/17937",
  "original_url": "https://arxiv.org/abs/2608.07419v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}