{
  "id": 16106,
  "url": "https://arxiv.org/abs/2608.02332v1",
  "title": "Diffusion Policy with Behavioral Advantage Correction for Offline Reinforcement Learning",
  "summary": "In offline reinforcement learning (RL), the distribution shift between behavioral data and the learned policy can lead to erroneous \\emph{Q}-value estimation, thereby misguiding the direction of policy optimization. To address this issue, we develop a behavioral advantage corrected policy evaluation (BAC-PE) approach, which utilizes the \\emph{Q}-function of the behavior policy to correct the learned policy's \\emph{Q}-function, thus mitigating pessimistic conservatism and overestimation bias. Fur",
  "authors": "Botao Dong, Longyang Huang, Ning Pang, Hongtian Chen",
  "category": "research",
  "topics": "bias-fairness,regulation",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-03T14:52:46.000Z",
  "fetched_at": "2026-08-04T05:10:21.797Z",
  "source_slug": "x-arxiv-cs-ai",
  "source_name": "arXiv cs.AI",
  "source_homepage": "https://arxiv.org/list/cs.AI/recent",
  "ethics_ai_record_url": "https://ethics.ai/record/16106",
  "original_url": "https://arxiv.org/abs/2608.02332v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}