{
  "id": 16934,
  "url": "https://arxiv.org/abs/2608.04794v1",
  "title": "Privileged, but Biased: How PI-Conditioned Teachers Break Self-Distillation",
  "summary": "Self-distillation (SD) has emerged as a compute-efficient alternative to reinforcement learning with verifiable rewards: a self-teacher, conditioned on privileged information (PI) about the answer such as a reference solution, supplies dense per-token supervision to a student that never sees it. Reported gains, however, come almost exclusively from narrow, low-difficulty settings, leaving open a basic question: as a lone objective, with no reward term, does SD teach anything? We reproduce SDPO's",
  "authors": "Sarthak Harne, Chinmay Karkar, Yash Pandya, Ahmed Awadallah, Akshay Nambi",
  "category": "research",
  "topics": "bias-fairness,children-education",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-05T12:59:10.000Z",
  "fetched_at": "2026-08-06T05:10:11.148Z",
  "source_slug": "x-arxiv-cs-ai",
  "source_name": "arXiv cs.AI",
  "source_homepage": "https://arxiv.org/list/cs.AI/recent",
  "ethics_ai_record_url": "https://ethics.ai/record/16934",
  "original_url": "https://arxiv.org/abs/2608.04794v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}