{
  "id": 14819,
  "url": "https://arxiv.org/abs/2607.26862v1",
  "title": "ReCo: Reweighting GRPO Against Distributional Concentration",
  "summary": "Group Relative Policy Optimization (GRPO) has become a standard reinforcement learning method for post-training language models. Recent work shows that GRPO can reduce the base model's reasoning capacity and underperform it in Pass@k when k is large, indicating reduced coverage of reasoning paths. We find that this reduction is associated with GRPO concentrating on responses that the base model already generates with high probability. We trace this concentration to two mechanisms in the GRPO upd",
  "authors": "Junoh Park, Junseo Hwang, Wonguk Cho, Taesup Kim",
  "category": "research",
  "topics": "regulation",
  "orgs": null,
  "regions": null,
  "published_at": "2026-07-29T12:45:55.000Z",
  "fetched_at": "2026-07-30T05:10:24.387Z",
  "source_slug": "x-arxiv-cs-ai",
  "source_name": "arXiv cs.AI",
  "source_homepage": "https://arxiv.org/list/cs.AI/recent",
  "ethics_ai_record_url": "https://ethics.ai/record/14819",
  "original_url": "https://arxiv.org/abs/2607.26862v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}