{
  "id": 176,
  "url": "https://arxiv.org/abs/2607.05804v1",
  "title": "TurnOPD: Making On-Policy Distillation Turn-Aware for Efficient Long-Horizon Agent Training",
  "summary": "On-policy distillation (OPD) trains a student policy by matching a stronger teacher on the student's own trajectories, offering a promising framework for language agent training. However, its application to long-horizon agentic tasks remains insufficiently explored. We identify two key inefficiencies in vanilla agent OPD: (1) full-horizon rollouts often waste wall-clock resources on tail turns that provide weak and noisy KL supervision, and (2) trajectory-level KL objectives concentrate most of ",
  "authors": "Yuhang Zhou, Kai Zheng, Haoling Li, Dengyun Peng, Can Xu, Jingjing Chen",
  "category": "research",
  "topics": "regulation,children-education,agents-autonomy",
  "orgs": null,
  "regions": null,
  "published_at": "2026-07-07T03:56:35.000Z",
  "fetched_at": "2026-07-14T14:14:19.971Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/176",
  "original_url": "https://arxiv.org/abs/2607.05804v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}