{
  "id": 3660,
  "url": "https://arxiv.org/abs/2605.27284v2",
  "title": "FineVLA: Fine-Grained Instruction Alignment for Steerable Vision-Language-Action Policies",
  "summary": "Vision-Language-Action (VLA) models are increasingly expected to not only complete robot tasks, but also follow human instructions about how those tasks should be executed. However, existing robot datasets usually pair trajectories with coarse goal-level language, leaving execution-critical details such as active arm, approach direction, and contact region unspecified. This limits steerable policy learning and robotic video understanding. We introduce FineVLA, an open framework for action-aligne",
  "authors": "Xintong Hu, Xuhong Huang, Jinyu Zhang, Yutong Yao, Yuchong Sun, Qiuyue Wang et al.",
  "category": "research",
  "topics": "regulation,safety-alignment,agents-autonomy",
  "orgs": null,
  "regions": null,
  "published_at": "2026-05-26T17:01:10.000Z",
  "fetched_at": "2026-07-14T16:30:23.248Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/3660",
  "original_url": "https://arxiv.org/abs/2605.27284v2",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}