{
  "id": 19139,
  "url": "https://arxiv.org/abs/2608.13489",
  "title": "DreamX-Phi 1.0: Action-Conditioned Video World Model for Robotic Manipulation",
  "summary": "We present DreamX-Phi 1.0, an action-conditioned video world model for robotic manipulation that, given an observed frame, a language instruction, and a prescribed action sequence comprising end-effector poses and gripper states, predicts the resulting future observations. Yet realism alone does not guarantee faithfulness: a convincing rollout can still move the wrong arm or lose the manipulated object. To ensure the prediction respects each arm's commanded path, we inject per-arm SE(3) transfor",
  "authors": "DreamX Team, Rui Chen, Xiangxiang Chu, Geng Li, Jifan Li, Qingfeng Shi",
  "category": "research",
  "topics": "agents-autonomy",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-12T20:00:00.000Z",
  "fetched_at": "2026-08-14T05:10:49.168Z",
  "source_slug": "hf-daily",
  "source_name": "HuggingFace Daily Papers",
  "source_homepage": "https://huggingface.co/papers",
  "ethics_ai_record_url": "https://ethics.ai/record/19139",
  "original_url": "https://arxiv.org/abs/2608.13489",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}