{
  "id": 14493,
  "url": "https://arxiv.org/abs/2607.23731v1",
  "title": "Outcome-Confounded Local Supervision in On-Policy Distillation",
  "summary": "On-policy distillation (OPD) trains a student on its own trajectories while a teacher supplies dense token-level likelihoods at student-visited prefixes. These likelihoods are often read locally: agreement appears safe to imitate, whereas disagreement appears to identify an error. We show that both readings are confounded by the outcome of the completed trajectory. We introduce an outcome-resolved diagnostic that crosses pointwise teacher-student divergence with final-answer correctness, separat",
  "authors": "Guoqing Ma",
  "category": "research",
  "topics": "regulation,healthcare,children-education",
  "orgs": null,
  "regions": null,
  "published_at": "2026-07-26T16:03:33.000Z",
  "fetched_at": "2026-07-29T05:10:12.205Z",
  "source_slug": "arxiv-cslg",
  "source_name": "arXiv cs.LG",
  "source_homepage": "https://arxiv.org/list/cs.LG/recent",
  "ethics_ai_record_url": "https://ethics.ai/record/14493",
  "original_url": "https://arxiv.org/abs/2607.23731v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}