{
  "id": 17365,
  "url": "https://arxiv.org/abs/2608.05730v1",
  "title": "SpaceVLA: Spatially Grounded VLA for Robotic Manipulation with User-Authored Grasp and Place Anchors",
  "summary": "Vision-language-action (VLA) models follow language commands but often lack explicit spatial intent for manipulation. We present Visual Intent Anchors, an XR pipeline that lets users specify grasp and placement regions and renders them as image-space overlays for VLA control. We collect 200 Unity pick-and-place demonstrations and fine-tune OpenVLA-7B with LoRA on temporally subsampled annotated observations. The policy predicts tokenized 7-DoF incremental actions from marked RGB observations and",
  "authors": "Daniia Zinniatullina, Iaroslav Kolomiets, Mikhail Konenkov, Miguel Altamirano Cabrera, Dzmitry Tsetserukou",
  "category": "research",
  "topics": "regulation,agents-autonomy",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-06T08:14:44.000Z",
  "fetched_at": "2026-08-07T05:10:58.501Z",
  "source_slug": "x-arxiv-cs-hc",
  "source_name": "arXiv cs.HC",
  "source_homepage": "https://arxiv.org/list/cs.HC/recent",
  "ethics_ai_record_url": "https://ethics.ai/record/17365",
  "original_url": "https://arxiv.org/abs/2608.05730v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}