{
  "id": 19475,
  "url": "https://arxiv.org/abs/2608.12957v1",
  "title": "I-SDPO: Instance-Level Adaptive Self-Distillation Policy Optimization",
  "summary": "Group Relative Policy Optimization (GRPO) learns from reward differences within a rollout group, but receives no useful relative signal when every sampled response is incorrect. Privileged self-distillation can fill this gap with dense token supervision, yet applying it throughout training creates a different failure mode: the teacher is a biased, low-variance surrogate for the reward objective, so persistent imitation can oppose reward-improving updates after the policy becomes capable of produ",
  "authors": "Yubo Zhang, Xinhong Ma, Zezhong Tan, Ziqiang Dong",
  "category": "research",
  "topics": "bias-fairness,regulation",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-13T08:37:24.000Z",
  "fetched_at": "2026-08-14T05:10:49.168Z",
  "source_slug": "arxiv-cslg",
  "source_name": "arXiv cs.LG",
  "source_homepage": "https://arxiv.org/list/cs.LG/recent",
  "ethics_ai_record_url": "https://ethics.ai/record/19475",
  "original_url": "https://arxiv.org/abs/2608.12957v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}