{
  "id": 1171,
  "url": "https://arxiv.org/abs/2606.11559v1",
  "title": "HERO: Hindsight-Enhanced Reflection from Environment Observations for Agentic Self-Distillation",
  "summary": "Reinforcement learning typically improves multi-turn agent capabilities through the terminal outcome of the trajectories, which makes it difficult to determine credit assignments for each intermediate turns. Recent on-policy self-distillation methods offer a promising alternative by converting privileged feedback into dense token-level supervision through a self-teacher. Our study is motivated by the unexpected performance degradation observed when naively extending this paradigm to multi-turn s",
  "authors": "Haoran Liu, Yuwei Zhang, Xiyao Li, Bohan Lyu, Jingbo Shang",
  "category": "research",
  "topics": "regulation,agents-autonomy,environment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-06-10T01:35:34.000Z",
  "fetched_at": "2026-07-14T14:15:03.616Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/1171",
  "original_url": "https://arxiv.org/abs/2606.11559v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}