{
  "id": 602,
  "url": "https://arxiv.org/abs/2606.26095v1",
  "title": "Learning Action Priors for Cross-embodiment Robot Manipulation",
  "summary": "Most Vision-Language-Action (VLA) models build on a Vision-Language Model (VLM) backbone by attaching an action module and optimizing the full policy jointly. This design inherits strong visual and linguistic priors from the VLM, but leaves the action module to learn physical motion almost from scratch. As a result, the policy lacks an explicit motion prior, forcing early optimization to simultaneously discover temporal action dynamics and cross-modal alignment, a challenge further amplified in ",
  "authors": "Dong Jing, Tianqi Zhang, Jiaqi Liu, Jinman Zhao, Zelong Sun, Li Erran Li et al.",
  "category": "research",
  "topics": "regulation,safety-alignment,agents-autonomy",
  "orgs": null,
  "regions": null,
  "published_at": "2026-06-24T17:59:56.000Z",
  "fetched_at": "2026-07-14T14:14:41.548Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/602",
  "original_url": "https://arxiv.org/abs/2606.26095v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}