{
  "id": 4847,
  "url": "https://arxiv.org/abs/2605.05995v2",
  "title": "Safety Anchor: Defending Harmful Fine-tuning via Geometric Bottlenecks",
  "summary": "The safety alignment of Large Language Models (LLMs) remains vulnerable to Harmful Fine-tuning (HFT). While existing defenses impose constraints on parameters, gradients, or internal representations, we observe that they can be effectively circumvented under persistent HFT. Our analysis traces this failure to the inherent redundancy of the high-dimensional parameter space: attackers exploit optimization trajectories that are orthogonal to defense constraints to restore harmful capabilities while",
  "authors": "Guoxin Lu, Letian Sha, Qing Wang, Peijie Sun, Hao Zhou, Hua Dai et al.",
  "category": "research",
  "topics": "safety-alignment,military-security",
  "orgs": null,
  "regions": null,
  "published_at": "2026-05-07T10:47:53.000Z",
  "fetched_at": "2026-07-14T16:31:17.583Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/4847",
  "original_url": "https://arxiv.org/abs/2605.05995v2",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}