{
  "id": 236,
  "url": "https://arxiv.org/abs/2607.04265v1",
  "title": "HALO-WA: Hybrid-Attention Latent-Guided Online Reinforcement Learning for World-Action Models",
  "summary": "World-action (WA) models can generate long-horizon action chunks for general-purpose robotic manipulation, but they remain vulnerable to calibration, perception, and contact-dynamics errors in real-world precision tasks, often failing in the final few millimeters of alignment or insertion. We propose HALO-WA, a hybrid-attention latent-guided online reinforcement learning (RL) framework for WA models, which leverages latent features and action priors from the WA generation process through a light",
  "authors": "Angen Ye, Weijie Ke, Xiaofeng Wang, Xinze Chen, Chaojun Ni, Guosheng Zhao et al.",
  "category": "research",
  "topics": "safety-alignment,agents-autonomy",
  "orgs": null,
  "regions": null,
  "published_at": "2026-07-05T12:24:14.000Z",
  "fetched_at": "2026-07-14T14:14:24.247Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/236",
  "original_url": "https://arxiv.org/abs/2607.04265v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}