{
  "id": 5860,
  "url": "https://arxiv.org/abs/2604.13016v2",
  "title": "Rethinking On-Policy Distillation of Large Language Models: Phenomenology, Mechanism, and Recipe",
  "summary": "On-policy distillation (OPD) has become a core technique in the post-training of large language models, yet its training dynamics remain poorly understood. This paper provides a systematic investigation of OPD dynamics and mechanisms. We first identify that two conditions govern whether OPD succeeds or fails: (i) the student and teacher should share compatible thinking patterns; and (ii) even with consistent thinking patterns and higher scores, the teacher must offer genuinely new capabilities b",
  "authors": "Yaxuan Li, Yuxin Zuo, Bingxiang He, Jinqian Zhang, Chaojun Xiao, Cheng Qian et al.",
  "category": "research",
  "topics": "regulation,children-education,finance-investment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-04-14T17:54:28.000Z",
  "fetched_at": "2026-07-14T16:32:02.062Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/5860",
  "original_url": "https://arxiv.org/abs/2604.13016v2",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}