{
  "id": 7630,
  "url": "https://arxiv.org/abs/2603.05772v2",
  "title": "Depth Charge: Jailbreak Large Language Models from Deep Safety Attention Heads",
  "summary": "Currently, open-sourced large language models (OSLLMs) have demonstrated remarkable generative performance. However, as their structure and weights are made public, they are exposed to jailbreak attacks even after alignment. Existing attacks operate primarily at shallow levels, such as the prompt or embedding level, and often fail to expose vulnerabilities rooted in deeper model components, which creates a false sense of security for successful defense. In this paper, we propose \\textbf{\\underli",
  "authors": "Jinman Wu, Yi Xie, Shiqian Zhao, Xiaofeng Chen",
  "category": "research",
  "topics": "safety-alignment,military-security",
  "orgs": null,
  "regions": null,
  "published_at": "2026-03-06T00:13:48.000Z",
  "fetched_at": "2026-07-14T16:33:21.050Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/7630",
  "original_url": "https://arxiv.org/abs/2603.05772v2",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}