{
  "id": 385,
  "url": "https://arxiv.org/abs/2606.31876v2",
  "title": "Harnessing Textual Refusal Directions for Multimodal Safety",
  "summary": "To improve safety in Large Language Models (LLMs) we can either perform post-training alignment or exploit refusal directions in the activation space. Both strategies are less feasible in Multimodal LLMs (MLLMs) as they require unsafe multimodal data, harder to collect than their unimodal counterpart. In this work, we relax this constraint and investigate whether textual refusal directions, extracted directly from the LLM backbone, generalize across modalities (i.e., image, video). Preliminary f",
  "authors": "Moreno D'Incà, Nicu Sebe, Massimiliano Mancini",
  "category": "research",
  "topics": "safety-alignment,finance-investment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-06-30T15:57:50.000Z",
  "fetched_at": "2026-07-14T14:14:28.438Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/385",
  "original_url": "https://arxiv.org/abs/2606.31876v2",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}