{
  "id": 12345,
  "url": "https://arxiv.org/abs/2607.18958v1",
  "title": "Dual Adversarial Fine-tuning for Enhancing Robustness of Large Vision Language Model",
  "summary": "While Large Vision-Language Models (LVLMs), represented by LLaVA and GPT-4V, have demonstrated remarkable capabilities, their visual inputs remain vulnerable to adversarial attacks, posing significant security risks. Existing defense methods predominantly target single-task scenarios (e.g., zero-shot classification) and consequently lack generalizability across various multimodal tasks. To address this limitation, we propose a dual adversarial fine-tuning framework that jointly optimizes visual",
  "authors": "Sibo Wang, Jie Zhang, Shiguang Shan, Xilin Chen, Wen Gao",
  "category": "research",
  "topics": "military-security",
  "orgs": null,
  "regions": null,
  "published_at": "2026-07-21T10:49:29.000Z",
  "fetched_at": "2026-07-22T05:10:49.469Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/12345",
  "original_url": "https://arxiv.org/abs/2607.18958v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}