{
  "id": 16237,
  "url": "https://arxiv.org/abs/2608.01735",
  "title": "DAPD: Dual-Anchored Policy Distillation",
  "summary": "On-policy (self) distillation (OPSD) is increasingly adopted for language-model post-training. It strengthens the teacher with privileged information but can induce a privilege illusion: the student learns privilege-dependent behavior it cannot reproduce from its inference-time context, yet behaves as if the training-time privileged information remained available, ultimately degrading performance. In this paper, we identify information asymmetry between the privileged teacher and the student at",
  "authors": "Jianyu Wu, Yizhou Wang, Encheng Su, Chen Tang, Shixiang Tang",
  "category": "research",
  "topics": "regulation,children-education",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-02T20:00:00.000Z",
  "fetched_at": "2026-08-05T05:10:44.550Z",
  "source_slug": "hf-daily",
  "source_name": "HuggingFace Daily Papers",
  "source_homepage": "https://huggingface.co/papers",
  "ethics_ai_record_url": "https://ethics.ai/record/16237",
  "original_url": "https://arxiv.org/abs/2608.01735",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}