{
  "id": 5873,
  "url": "https://arxiv.org/abs/2604.12617v2",
  "title": "SOAR: Self-Correction for Optimal Alignment and Refinement in Diffusion Models",
  "summary": "The post-training pipeline for diffusion models currently has two stages: supervised fine-tuning (SFT) on curated data and reinforcement learning (RL) with reward models. A fundamental gap separates them. SFT optimizes the denoiser only on ground-truth states sampled from the forward noising process; once inference deviates from these ideal states, subsequent denoising relies on out-of-distribution generalization rather than learned correction, exhibiting the same exposure bias that afflicts aut",
  "authors": "You Qin, Linqing Wang, Hao Fei, Roger Zimmermann, Liefeng Bo, Qinglin Lu et al.",
  "category": "research",
  "topics": "bias-fairness,safety-alignment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-04-14T11:45:15.000Z",
  "fetched_at": "2026-07-14T16:32:06.466Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/5873",
  "original_url": "https://arxiv.org/abs/2604.12617v2",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}