{
  "id": 16101,
  "url": "https://arxiv.org/abs/2608.02365v1",
  "title": "Faster-WAM: Do World Action Models Need Deep Action Modules?",
  "summary": "World Action Models (WAMs) couple robot action prediction with video world models. Existing WAMs with shared-backbone and Mixture-of-Transformers designs generally tie the depth of the action module to that of the video backbone, resulting in substantial computational overhead and high inference latency. To address this limitation, we introduce Dock of Transformer (DoT), a video-centric design principle that treats a pretrained video Transformer as a representation hub and connects lightweight o",
  "authors": "Liheng Ma, Rui Heng Yang, Zhanguang Zhang, Mateo Clemente, Ziwen Hu, Tongtong Cao, Yingxue Zhang",
  "category": "research",
  "topics": "agents-autonomy",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-03T15:11:21.000Z",
  "fetched_at": "2026-08-04T05:10:21.797Z",
  "source_slug": "x-arxiv-cs-ai",
  "source_name": "arXiv cs.AI",
  "source_homepage": "https://arxiv.org/list/cs.AI/recent",
  "ethics_ai_record_url": "https://ethics.ai/record/16101",
  "original_url": "https://arxiv.org/abs/2608.02365v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}