{
  "id": 6459,
  "url": "https://arxiv.org/abs/2604.01723v1",
  "title": "Causal Scene Narration with Runtime Safety Supervision for Vision-Language-Action Driving",
  "summary": "Vision-Language-Action (VLA) models for autonomous driving must integrate diverse textual inputs, including navigation commands, hazard warnings, and traffic state descriptions, yet current systems often present these as disconnected fragments, forcing the model to discover on its own which environmental constraints are relevant to the current maneuver. We introduce Causal Scene Narration (CSN), which restructures VLA text inputs through intent-constraint alignment, quantitative grounding, and s",
  "authors": "Yun Li, Yidu Zhang, Simon Thompson, Ehsan Javanmardi, Manabu Tsukada",
  "category": "research",
  "topics": "safety-alignment,environment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-04-02T07:43:05.000Z",
  "fetched_at": "2026-07-14T16:32:28.613Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/6459",
  "original_url": "https://arxiv.org/abs/2604.01723v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}