{
  "id": 3630,
  "url": "https://arxiv.org/abs/2605.27846v1",
  "title": "EAPO: Entropy-Driven Adaptive Positive-Negative Sample Weighting for Policy Optimization in Open-Ended QA",
  "summary": "Large Reasoning Models are typically trained via reinforcement learning from verifiable rewards (RLVR). However, existing approaches adopt fixed weights for positive and negative samples, and the conclusions hardly generalize to open-ended question answering (QA). In this paper, we systematically investigate the roles of positive and negative samples in reinforcement learning for open-ended QA. We propose a reward-mean-based strategy for distinguishing positive from negative samples, and observe",
  "authors": "Yunsheng Zeng, Gen Li, Yuwei Miao, Xiandong Li, Yujin Wang, Siyu Chen et al.",
  "category": "research",
  "topics": "regulation,finance-investment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-05-27T02:04:00.000Z",
  "fetched_at": "2026-07-14T16:30:23.247Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/3630",
  "original_url": "https://arxiv.org/abs/2605.27846v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}