{
  "id": 718,
  "url": "https://arxiv.org/abs/2606.28385v1",
  "title": "RoboGaze: Evaluating Robot World Models via Structured Vision-Language Analysis",
  "summary": "Recent advances in robot world models enable synthetic video generation for embodied prediction and planning. However, evaluating these videos is challenging: visually realistic outputs often violate physical laws, temporal consistency, or task logic, while conventional metrics and monolithic Vision-Language Model (VLM) judges fail to generalize or provide precise diagnostic value. We present RoboGaze, a training-free, multi-agent VLM framework that provides structured, interpretable evaluation ",
  "authors": "Minh-Loi Nguyen, Nghiem Tuong Diep, Hung Khang Nguyen, Minh Le, Doanh Le Thien, Hoang H. Tran et al.",
  "category": "research",
  "topics": "healthcare,agents-autonomy",
  "orgs": null,
  "regions": null,
  "published_at": "2026-06-22T06:45:09.000Z",
  "fetched_at": "2026-07-14T14:14:46.033Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/718",
  "original_url": "https://arxiv.org/abs/2606.28385v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}