{
  "id": 14484,
  "url": "https://arxiv.org/abs/2607.24900v1",
  "title": "Inverse RL Helps Align AI by Imitating Humans",
  "summary": "Language model alignment aims to make model behavior reliably reflect desirable properties such as helpfulness, safety, and instruction following. Current approaches typically use supervised fine-tuning on demonstrations or reinforcement learning with rewards derived from verifiers or human feedback. These paradigms leave an important question underexplored: can demonstrations alone yield an implicit reward that can be inspected, reused, and optimized on-policy to align AI? Motivated by inverse",
  "authors": "Michał Wiliński, Liu Leqi, Chirag Nagpal",
  "category": "research",
  "topics": "regulation,safety-alignment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-07-27T16:45:31.000Z",
  "fetched_at": "2026-07-29T05:10:12.205Z",
  "source_slug": "arxiv-cslg",
  "source_name": "arXiv cs.LG",
  "source_homepage": "https://arxiv.org/list/cs.LG/recent",
  "ethics_ai_record_url": "https://ethics.ai/record/14484",
  "original_url": "https://arxiv.org/abs/2607.24900v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}