{
  "id": 18426,
  "url": "https://arxiv.org/abs/2608.10484v1",
  "title": "Lost in Reconstruction: Aligning Action Representations with Language in Vision-Language-Action Models",
  "summary": "Action verbs describe not only the physical outcomes of actions, but also how those actions are performed. Yet action representations in vision-language-action models (VLAs) are typically optimized for reconstruction under L1/L2 losses in raw action space, where numerical proximity need not reflect linguistically meaningful distinctions. On BridgeV2, we show that action trajectories contain verb-grounding information beyond visual state changes, and that reconstruction-only discrete tokenization",
  "authors": "Li Wenjie, Yash Jangir, Ignacy Stepka, Yash Agarwal, Marion Kipsang, Yonatan Bisk",
  "category": "research",
  "topics": null,
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-11T04:57:17.000Z",
  "fetched_at": "2026-08-12T05:10:43.828Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/18426",
  "original_url": "https://arxiv.org/abs/2608.10484v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}