{
  "id": 6523,
  "url": "https://arxiv.org/abs/2604.00223v1",
  "title": "Diversity-Aware Reverse Kullback-Leibler Divergence for Large Language Model Distillation",
  "summary": "Reverse Kullback-Leibler (RKL) divergence has recently emerged as the preferred objective for large language model (LLM) distillation, consistently outperforming forward KL (FKL), particularly in regimes with large vocabularies and significant teacher-student capacity mismatch, where RKL focuses learning on dominant modes rather than enforcing dense alignment. However, RKL introduces a structural limitation that drives the student toward overconfident predictions. We first provide an analysis of",
  "authors": "Hoang-Chau Luong, Dat Ba Tran, Lingwei Chen",
  "category": "research",
  "topics": "safety-alignment,children-education",
  "orgs": null,
  "regions": null,
  "published_at": "2026-03-31T20:39:47.000Z",
  "fetched_at": "2026-07-14T16:32:33.101Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/6523",
  "original_url": "https://arxiv.org/abs/2604.00223v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}