{
  "id": 3538,
  "url": "https://arxiv.org/abs/2605.29398v1",
  "title": "GDSD: Reinforcement Learning as Guided Denoiser Self-Distillation for Diffusion Language Models",
  "summary": "Reinforcement learning (RL) can be used to improve the policy (denoiser) of diffusion large language models (dLLMs), while being hindered by the intractability of the policy likelihood. A dominant and efficient family of methods replaces the likelihood in standard RL with its evidence lower bound (ELBO), estimated from randomly masked sequences. Despite being well aligned with pre-training, these approaches introduce bias through training--inference mismatch by using the ELBO as a likelihood sur",
  "authors": "Xiaohang Tang, Keyue Jiang, Che Liu, Qifang Zhao, Xiaoxiao Xu, Sangwoong Yoon et al.",
  "category": "research",
  "topics": "bias-fairness,regulation",
  "orgs": null,
  "regions": null,
  "published_at": "2026-05-28T05:47:40.000Z",
  "fetched_at": "2026-07-14T16:30:18.857Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/3538",
  "original_url": "https://arxiv.org/abs/2605.29398v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}