{
  "id": 6515,
  "url": "https://arxiv.org/abs/2604.00310v1",
  "title": "Robust Multimodal Safety via Conditional Decoding",
  "summary": "Multimodal large-language models (MLLMs) often experience degraded safety alignment when harmful queries exploit cross-modal interactions. Models aligned on text alone show a higher rate of successful attacks when extended to two or more modalities. In this work, we propose a simple conditional decoding strategy, CASA (Classification Augmented with Safety Attention) that utilizes internal representations of MLLMs to predict a binary safety token before response generation. We introduce a novel s",
  "authors": "Anurag Kumar, Raghuveer Peri, Jon Burnsky, Alexandru Nelus, Rohit Paturi, Srikanth Vishnubhotla et al.",
  "category": "research",
  "topics": "safety-alignment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-03-31T23:19:50.000Z",
  "fetched_at": "2026-07-14T16:32:33.100Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/6515",
  "original_url": "https://arxiv.org/abs/2604.00310v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}