{
  "id": 4155,
  "url": "https://arxiv.org/abs/2605.20246v2",
  "title": "GROW: Aligning GRPO with State-Action Modeling for Open-World VLM Agents",
  "summary": "Recently, vision-language model (VLM) agents have shown promising progress in open-world tasks, where successful task completion often requires multiple turns of visual perception and action execution. However, existing methods still rely primarily on Supervised Fine-Tuning (SFT) with expert demonstrations, while the advanced reinforcement learning (RL) algorithm, specifically Group Relative Policy Optimization (GRPO), has not been effectively employed for multi-turn RL in these tasks because st",
  "authors": "Xiongbin Wu, Zhihao Luo, Shanzhe Lei, Lechao Zhang, Xuhong Wang, Jie Yang et al.",
  "category": "research",
  "topics": "regulation,agents-autonomy",
  "orgs": null,
  "regions": null,
  "published_at": "2026-05-18T04:56:59.000Z",
  "fetched_at": "2026-07-14T16:30:45.941Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/4155",
  "original_url": "https://arxiv.org/abs/2605.20246v2",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}