{
  "id": 5127,
  "url": "https://arxiv.org/abs/2605.00425v3",
  "title": "AEM: Adaptive Entropy Modulation for Multi-Turn Agentic Reinforcement Learning",
  "summary": "Reinforcement learning (RL) has substantially improved the ability of large language model (LLM) agents to interact with environments and solve multi-turn tasks. However, effective agentic RL remains challenging: sparse outcome-only rewards provide limited guidance for assigning credit to individual steps within long interaction trajectories. Existing approaches often introduce dense intermediate supervision, such as process reward models or auxiliary self-supervised signals, which increases sup",
  "authors": "Haotian Zhao, Songlin Zhou, Yuxin Zhang, Stephen S. -T. Yau, Wenyu Zhang, Lun Tian et al.",
  "category": "research",
  "topics": "agents-autonomy,environment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-05-01T05:54:37.000Z",
  "fetched_at": "2026-07-14T16:31:31.212Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/5127",
  "original_url": "https://arxiv.org/abs/2605.00425v3",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}