{
  "id": 14131,
  "url": "https://arxiv.org/abs/2607.25912v1",
  "title": "SAM3D-Guided Object-Centric Representation Alignment for Vision-Language-Action Models",
  "summary": "Vision-Language-Action (VLA) models have shown strong potential for general robot manipulation, but most existing models rely on 2D visual-language backbones and lack fine-grained 3D understanding of target objects, especially under occlusion, pose variation, scale changes, and precise spatial interaction. We propose an object-centric 3D representation alignment framework built upon $π_0$, using SAM3D as a frozen 3D teacher to provide target-object 3D priors during training. Specifically, we loc",
  "authors": "Zonghe Liu, Shanyuan Jie, Xiaoquan Sun, Chen Cao, Zetian Xu, Zongsheng Liu et al.",
  "category": "research",
  "topics": "safety-alignment,agents-autonomy",
  "orgs": null,
  "regions": null,
  "published_at": "2026-07-28T16:05:32.000Z",
  "fetched_at": "2026-07-29T05:10:12.205Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/14131",
  "original_url": "https://arxiv.org/abs/2607.25912v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}