{
  "id": 10201,
  "url": "https://arxiv.org/abs/2607.11475v1",
  "title": "HyperSafe: Inference-Time Safety Recovery for Fine-Tuned Language Models",
  "summary": "Safety alignment in large language models can be fragile under fine-tuning, as even benign task adaptation may increase harmful compliance. Existing defenses mainly follow two directions: they either intervene during or after fine-tuning through retraining or weight modification, which can be costly and may hurt task performance, or they use model-agnostic safety classifiers, which may miss failures specific to a given fine-tuned checkpoint. These limitations motivate a post hoc, model-specific,",
  "authors": "Aznaur Aliev, Carlos Hinojosa, Abdelrahman Eldesokey, Bang An, Bernard Ghanem, Yibo Yang",
  "category": "research",
  "topics": "regulation,safety-alignment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-07-13T12:28:43.000Z",
  "fetched_at": "2026-07-14T16:55:59.928Z",
  "source_slug": "arxiv-cslg",
  "source_name": "arXiv cs.LG",
  "source_homepage": "https://arxiv.org/list/cs.LG/recent",
  "ethics_ai_record_url": "https://ethics.ai/record/10201",
  "original_url": "https://arxiv.org/abs/2607.11475v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}