{
  "id": 4723,
  "url": "https://arxiv.org/abs/2605.07804v3",
  "title": "Prune-OPD: Efficient and Reliable On-Policy Distillation for Long-Horizon Reasoning",
  "summary": "On-policy distillation (OPD) leverages dense teacher rewards to enhance reasoning models. However, scaling OPD to long-horizon tasks exposes a critical flaw: as the student's generated prefix inevitably diverges from the teacher's thought process, the teacher's dense reward loses local exploitability. Continuing to generate and evaluate tokens on these ``drifted'' trajectories not only degrades reward quality but also incurs massive computational waste. To address this, we introduce \\textbf{Prun",
  "authors": "Zhicheng Yang, Zhijiang Guo, Yifan Song, Minrui Xu, Yongxin Wang, Yiwei Wang et al.",
  "category": "research",
  "topics": "regulation,children-education",
  "orgs": null,
  "regions": null,
  "published_at": "2026-05-08T14:38:53.000Z",
  "fetched_at": "2026-07-14T16:31:12.746Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/4723",
  "original_url": "https://arxiv.org/abs/2605.07804v3",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}