{
  "id": 17983,
  "url": "https://arxiv.org/abs/2608.06756",
  "title": "Capek 0.5: An Execution-Centric Vision-Language Model for Embodied Intelligence",
  "summary": "Vision-language models are increasingly serving as the reasoning core of embodied agents. Robot execution is inherently iterative: each action reshapes the scene and physical state, continually renewing what must be perceived, reasoned about, and verified. Meeting these demands requires complementary capabilities that differ in supervision signals, prediction formats, and verification criteria. Existing approaches typically develop these capabilities against isolated, task-specific objectives, l",
  "authors": "Ying Chen, Weizhen Li, Zhe Hu, Zhenjiang Li, Rui Jiang, Zhifeng Gu",
  "category": "research",
  "topics": "agents-autonomy",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-06T20:00:00.000Z",
  "fetched_at": "2026-08-11T05:10:37.351Z",
  "source_slug": "hf-daily",
  "source_name": "HuggingFace Daily Papers",
  "source_homepage": "https://huggingface.co/papers",
  "ethics_ai_record_url": "https://ethics.ai/record/17983",
  "original_url": "https://arxiv.org/abs/2608.06756",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}