{
  "id": 207,
  "url": "https://arxiv.org/abs/2607.04728v1",
  "title": "Turning Off-Policy Tokens On-Policy: A Plug-in Approach for Improving LLM Alignment",
  "summary": "Reinforcement learning (RL) post-training for large language models (LLMs) follows a efficient paradigm of \"rollout then update\", which inevitably results in off-policy training data. To resolve this, Importance sampling (IS) is proposed, while the token-level ratios compound over long sequences, causing severe variance exploded. A natural idea is \"transferring\" these off-policy token into on-policy token, so that the importance scores for correction are unnecessary. Following this idea, we prop",
  "authors": "Yu Li, Xiuyu Li, Mingyang Yi, Jiaxing Wang, zhangliangxu, Zhaolong Xing et al.",
  "category": "research",
  "topics": "regulation,safety-alignment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-07-06T06:59:44.000Z",
  "fetched_at": "2026-07-14T14:14:24.245Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/207",
  "original_url": "https://arxiv.org/abs/2607.04728v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}