{
  "id": 18336,
  "url": "https://arxiv.org/abs/2608.07935v1",
  "title": "Adaptive Supervised Anchoring for On-Policy Self-Distillation",
  "summary": "On-policy self-distillation (OPSD) adapts a language model by distilling guidance from a frozen teacher on trajectories sampled from the student. Its effectiveness, however, depends critically on the quality of those trajectories. We show that when student rollouts drift from target trajectories, conditioning the teacher on off-target prefixes substantially weakens its task-relevant supervision. Controlled prefix-corruption experiments expose this failure mode, which we term rollout-conditioned",
  "authors": "Meilin Yang, Zixuan Ding, Jianhao Nie, Weite Zhang, Yuxin Zhang, Zhiming Shao et al.",
  "category": "research",
  "topics": "regulation,children-education",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-08T05:46:32.000Z",
  "fetched_at": "2026-08-11T05:10:37.351Z",
  "source_slug": "arxiv-cslg",
  "source_name": "arXiv cs.LG",
  "source_homepage": "https://arxiv.org/list/cs.LG/recent",
  "ethics_ai_record_url": "https://ethics.ai/record/18336",
  "original_url": "https://arxiv.org/abs/2608.07935v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}