{
  "id": 16599,
  "url": "https://arxiv.org/abs/2608.00782",
  "title": "Distill Where You Fail: Recovering Learning Signals of Negative RL-Groups from Adaptive Teacher Guidance",
  "summary": "Reinforcement learning with verifiable rewards (RLVR) has become a standard paradigm for post-training large language models (LLMs). While Group Relative Policy Optimization (GRPO) is widely adopted, it suffers from sparse reward signals and loses gradients entirely when all responses within a group receive identical rewards. On-policy distillation (OPD) offers a natural remedy by providing dense, token-level supervision from a teacher model. However, naively combining GRPO with OPD leads to deg",
  "authors": "Zhuowen Han, Jinwei Xiao, Zhengxi Lu, Renren Jin, Zhiyuan Yao, Yuxin Liu",
  "category": "research",
  "topics": "regulation",
  "orgs": null,
  "regions": null,
  "published_at": "2026-07-31T20:00:00.000Z",
  "fetched_at": "2026-08-06T05:10:11.148Z",
  "source_slug": "hf-daily",
  "source_name": "HuggingFace Daily Papers",
  "source_homepage": "https://huggingface.co/papers",
  "ethics_ai_record_url": "https://ethics.ai/record/16599",
  "original_url": "https://arxiv.org/abs/2608.00782",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}