{
  "id": 4419,
  "url": "https://arxiv.org/abs/2605.12969v3",
  "title": "Revisiting Reinforcement Learning with Verifiable Rewards from a Contrastive Perspective",
  "summary": "Group Relative Policy Optimization (GRPO) is one of the most widely adopted RLVR algorithms for post-training large language models on reasoning tasks. We first show that GRPO admits an equivalent discriminative reformulation, in which policy optimization maximizes the expected score gap between verified positive and negative rollouts. This reformulation reveals two objective-level limitations: likelihood-misaligned surrogate scores, in which clipped ratio-based scores are optimized rather than ",
  "authors": "Feng Zhang, Xinhong Ma, Ziqiang Dong, Xi Leng, Jianfei Zhao, Xin Sun et al.",
  "category": "research",
  "topics": "bias-fairness,regulation,safety-alignment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-05-13T04:02:36.000Z",
  "fetched_at": "2026-07-14T16:30:59.237Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/4419",
  "original_url": "https://arxiv.org/abs/2605.12969v3",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}