{
  "id": 6129,
  "url": "https://arxiv.org/abs/2604.08178v2",
  "title": "Aligning Agents via Planning: A Benchmark for Trajectory-Level Reward Modeling",
  "summary": "In classical Reinforcement Learning from Human Feedback (RLHF), Reward Models (RMs) serve as the fundamental signal provider for model alignment. As Large Language Models evolve into agentic systems capable of autonomous tool invocation and complex reasoning, the paradigm of reward modeling faces unprecedented challenges -- most notably, the lack of benchmarks specifically designed to assess RM capabilities within tool-integrated environments. To address this gap, we present Plan-RewardBench, a ",
  "authors": "Jiaxuan Wang, Yulan Hu, Wenjin Yang, Zheng Pan, Xin Li, Lan-Zhe Guo",
  "category": "research",
  "topics": "safety-alignment,agents-autonomy,environment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-04-09T12:35:06.000Z",
  "fetched_at": "2026-07-14T16:32:15.637Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/6129",
  "original_url": "https://arxiv.org/abs/2604.08178v2",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}