{
  "id": 15903,
  "url": "https://arxiv.org/abs/2608.01735v1",
  "title": "DAPD: Dual-Anchored Policy Distillation",
  "summary": "On-policy (self) distillation (OPSD) is increasingly adopted for language-model post-training. It strengthens the teacher with privileged information but can induce a privilege illusion: the student learns privilege-dependent behavior it cannot reproduce from its inference-time context, yet behaves as if the training-time privileged information remained available, ultimately degrading performance. In this paper, we identify information asymmetry between the privileged teacher and the student at",
  "authors": "Jianyu Wu, Yizhou Wang, Encheng Su, Chen Tang, Shixiang Tang",
  "category": "research",
  "topics": "regulation,children-education",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-03T06:00:44.000Z",
  "fetched_at": "2026-08-04T05:10:21.797Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/15903",
  "original_url": "https://arxiv.org/abs/2608.01735v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}