{
  "id": 13020,
  "url": "https://arxiv.org/abs/2607.10848",
  "title": "Predictive Divergence Masks for LLM RL",
  "summary": "Reinforcement learning for large language models (LLMs) typically relies on trust-region masks to stabilize off-policy updates. The dominant PPO-style approach uses the sampled-token importance ratio for two criteria: a proximity criterion, which asks whether the policy has moved too far from the behavior policy, and a direction criterion, which asks whether the update pushes it farther away. Recent work DPPO improves the proximity criterion by replacing PPO's ratio-based test with a probability",
  "authors": "Xiangxin Zhou, Jiarui Yao, Penghui Qi, Bowen Ping, Jiaqi Tang, Haonan Wang",
  "category": "research",
  "topics": "regulation",
  "orgs": null,
  "regions": null,
  "published_at": "2026-07-11T20:00:00.000Z",
  "fetched_at": "2026-07-25T05:10:48.796Z",
  "source_slug": "hf-daily",
  "source_name": "HuggingFace Daily Papers",
  "source_homepage": "https://huggingface.co/papers",
  "ethics_ai_record_url": "https://ethics.ai/record/13020",
  "original_url": "https://arxiv.org/abs/2607.10848",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}