{
  "id": 16981,
  "url": "https://arxiv.org/abs/2608.03929v2",
  "title": "Latent Reward Registers for Diffusion Preference Alignment",
  "summary": "Aligning diffusion models with human preferences usually relies on a sparse terminal reward evaluated on the final generated samples, presenting a severe temporal credit-assignment challenge across the multi-step denoising process. We propose Latent Reward Registers, a mechanism that estimates terminal preference directly from intermediate noisy latents by prepending learnable, position-free register tokens to the input sequence of a frozen Diffusion Transformer (DiT). This independent readout m",
  "authors": "Yuanshen Guan, Zipeng Feng, Chengru Song, Zhiwei Xiong, Peiqin Sun",
  "category": "research",
  "topics": "safety-alignment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-04T17:00:52.000Z",
  "fetched_at": "2026-08-06T05:10:11.148Z",
  "source_slug": "arxiv-cslg",
  "source_name": "arXiv cs.LG",
  "source_homepage": "https://arxiv.org/list/cs.LG/recent",
  "ethics_ai_record_url": "https://ethics.ai/record/16981",
  "original_url": "https://arxiv.org/abs/2608.03929v2",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}