{
  "id": 18028,
  "url": "https://arxiv.org/abs/2608.09123v1",
  "title": "RISE-RL: Rubric-Informed Selective Exploration for Open-Ended Reinforcement Learning",
  "summary": "Aligning Large Language Models (LLMs) for open-ended tasks is challenging because responses must satisfy multidimensional criteria without following a single correct generation trajectory. Existing rubric-based reinforcement learning (RL) methods compress fine-grained criterion-level feedback into scalar rewards, making persistent capability gaps difficult to target under limited on-policy exploration. We propose $\\textbf{RISE-RL}$ (Rubric-Informed Selective Exploration), which uses repeatedly m",
  "authors": "Jinkun Hou, Zhuo Liu, Huimin Ren, Hongsheng Xin, Pan Zhou, Kun Zhan",
  "category": "research",
  "topics": "regulation",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-10T05:02:28.000Z",
  "fetched_at": "2026-08-11T05:10:37.351Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/18028",
  "original_url": "https://arxiv.org/abs/2608.09123v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}