{
  "id": 5045,
  "url": "https://arxiv.org/abs/2605.01913v1",
  "title": "RefusalGuard: Geometry-Preserving Fine-Tuning for Safety in LLMs",
  "summary": "Fine-tuning safety-aligned language models for downstream tasks often leads to substantial degradation of refusal behavior, making models vulnerable to adversarial misuse. While prior work has shown that safety-relevant features are encoded in structured representations within the model's activation space, how these representations change during fine-tuning and why alignment degrades remains poorly understood. In this work, we investigate the representation-level mechanisms underlying alignment ",
  "authors": "Sadia Asif, Mohammad Mohammadi Amiri",
  "category": "research",
  "topics": "safety-alignment,finance-investment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-05-03T14:48:18.000Z",
  "fetched_at": "2026-07-14T16:31:26.338Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/5045",
  "original_url": "https://arxiv.org/abs/2605.01913v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}