{
  "id": 16656,
  "url": "https://arxiv.org/abs/2608.04788v1",
  "title": "Agentic Reinforcement Learning with Observation-Calibrated Self-Distillation",
  "summary": "Large language model agents are commonly trained through reinforcement learning with sparse trajectory-level rewards, which offer limited guidance on how strongly individual tokens should be updated. On-Policy Self-Distillation (OPSD) addresses this by re-scoring generated tokens under a privileged replay view to obtain dense, token-level supervision. However, we identify a confounding issue: the resulting support may reflect both the privileged information contained in the replay view and score",
  "authors": "Yi Yang, Cong Qin, Xiaodan Liu, Chishui Chen, Qing Dong, Yan Zhang et al.",
  "category": "research",
  "topics": "regulation,agents-autonomy",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-05T12:52:14.000Z",
  "fetched_at": "2026-08-06T05:10:11.148Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/16656",
  "original_url": "https://arxiv.org/abs/2608.04788v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}