{
  "id": 17336,
  "url": "https://arxiv.org/abs/2608.06243v1",
  "title": "DASH: Divergence-Adaptive Supervision Horizons for On-Policy Self-Distillation of Reasoning Models",
  "summary": "Reinforcement learning with verifiable rewards (RLVR) improves the reasoning capabilities of large language models using automatically verifiable outcome signals, but these signals are typically sparse and at the sequence-level. On-policy self-distillation (OPSD) mitigates this sparsity by querying a privileged teacher at student-visited prefixes and providing dense token-level distributional supervision. Although this dense supervision alleviates signal sparsity, we find that standard OPSD stil",
  "authors": "ZhiYan Hou, Xinyu Tang, Hongyan An, Jianjin Zhang, Weizhen Wang, Yunyun Han, Gengsheng Li, Xiangzhao Hao, Haiyun Guo, Wenbin Hu, Jinqiao Wang, Yafeng Deng",
  "category": "research",
  "topics": "regulation,children-education",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-06T16:29:24.000Z",
  "fetched_at": "2026-08-07T05:10:58.501Z",
  "source_slug": "x-arxiv-cs-ai",
  "source_name": "arXiv cs.AI",
  "source_homepage": "https://arxiv.org/list/cs.AI/recent",
  "ethics_ai_record_url": "https://ethics.ai/record/17336",
  "original_url": "https://arxiv.org/abs/2608.06243v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}