{
  "id": 285,
  "url": "https://arxiv.org/abs/2607.03126v2",
  "title": "ACPO: Adaptive Credit Policy Optimization via Fine-Grained Surrogate Entropy",
  "summary": "Reinforcement Learning (RL) has substantially improved the reasoning ability of large language models (LLMs), but sparse outcome rewards still make token-level credit assignment difficult. Existing scalable RL methods typically assign trajectory-level rewards uniformly across tokens, while recent entropy-aware approaches either rely on coarse detached heuristics or directly optimize true entropy, which can introduce non-local gradient components misaligned with sampled-token policy updates. We p",
  "authors": "Zijun Xie, Yuyang You, Yongzhi Li, Enlei Gong, Zeyu Chen, Quan Chen et al.",
  "category": "research",
  "topics": "regulation,safety-alignment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-07-03T09:14:27.000Z",
  "fetched_at": "2026-07-14T14:14:24.248Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/285",
  "original_url": "https://arxiv.org/abs/2607.03126v2",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}