{
  "id": 13612,
  "url": "https://arxiv.org/abs/2607.22393v1",
  "title": "SceneActBench: Can Agents Act on the 3D Scenes They See?",
  "summary": "Vision-language model (VLM) agents increasingly use tools to act on 3D scenes rather than only describe them. Existing 3D benchmarks score textual responses or single-object operations, leaving agent action on complete multi-object 3D scenes under evaluated. We present SceneActBench, a benchmark for visually conditioned action across five 3D tasks under a unified agent-environment loop. Given PNG images or sampled video frames and, where applicable, supplied 3D assets, an agent acts on a 3D envi",
  "authors": "Yifei Zhao, Xiangxin Zhou, Wenhao Yang, Jiaqi Tang, Pu Jian, Huanjin Yao et al.",
  "category": "research",
  "topics": "agents-autonomy,environment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-07-24T15:16:47.000Z",
  "fetched_at": "2026-07-27T05:10:06.638Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/13612",
  "original_url": "https://arxiv.org/abs/2607.22393v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}