{
  "id": 5753,
  "url": "https://arxiv.org/abs/2604.15577v1",
  "title": "Reward Weighted Classifier-Free Guidance as Policy Improvement in Autoregressive Models",
  "summary": "Consider an auto-regressive model that produces outputs x (e.g., answers to questions, molecules) each of which can be summarized by an attribute vector y (e.g., helpfulness vs. harmlessness, or bio-availability vs. lipophilicity). An arbitrary reward function r(y) encodes tradeoffs between these properties. Typically, tilting the model's sampling distribution to increase this reward is done at training time via reinforcement learning. However, if the reward function changes, re-alignment requir",
  "authors": "Alexander Peysakhovich, William Berman",
  "category": "research",
  "topics": "regulation,safety-alignment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-04-16T23:13:22.000Z",
  "fetched_at": "2026-07-14T16:31:57.535Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/5753",
  "original_url": "https://arxiv.org/abs/2604.15577v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}