{
  "id": 473,
  "url": "https://arxiv.org/abs/2606.29340v1",
  "title": "PHF: Privileged Hidden Flow for On-Policy Self-Distillation",
  "summary": "On-policy self-distillation (OPSD) trains a reasoning model on rollouts sampled from its own policy by matching a privileged teacher that also sees verified reference solutions. Existing OPSD objectives supervise only the output distribution, so privileged context affects training through a token-level divergence without directly supervising the internal computation that produced that distribution. We propose Privileged Hidden Flow (PHF), which additionally distills how a privileged teacher's hi",
  "authors": "Yuhan Li, Mingxu Zhang, Dazhong Shen, Ying Sun",
  "category": "research",
  "topics": "regulation",
  "orgs": null,
  "regions": null,
  "published_at": "2026-06-28T11:15:26.000Z",
  "fetched_at": "2026-07-14T14:14:32.649Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/473",
  "original_url": "https://arxiv.org/abs/2606.29340v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}