{
  "id": 18423,
  "url": "https://arxiv.org/abs/2608.10513v1",
  "title": "SafeCap: Improving LVLM Safety with Image Captioning Reinforcement Learning",
  "summary": "Large vision-language models (LVLMs) remain vulnerable to jailbreak attacks that exploit visual inputs to bypass safety alignment inherited from their language backbones. We propose SafeCap, a reinforcement-learning framework that aligns LVLMs through learned self-captioning. SafeCap trains a policy model to first generate a safety-relevant image caption and then produce a final answer; the caption is further optimized by whether it enables a frozen LLM to reach a safety-aligned decision. This c",
  "authors": "Caoyuan Ma, Wenpu Liu, Weichu Xie, Tian Gu, Shilei Zhao, Lingxi Min et al.",
  "category": "research",
  "topics": "regulation,safety-alignment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-11T05:37:59.000Z",
  "fetched_at": "2026-08-12T05:10:43.828Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/18423",
  "original_url": "https://arxiv.org/abs/2608.10513v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}