{
  "id": 728,
  "url": "https://arxiv.org/abs/2606.22676v1",
  "title": "Skin-Deep: A Geometric Diagnostic for Alignment Fragility in Large Language Model Representations",
  "summary": "Alignment tuning is meant to make harmful-request refusal robust, yet this safety behavior can be erased by a small set of benign fine-tuning examples. This is a deployment risk for open-weight models because a checkpoint can pass refusal tests at release time and later lose refusal under low-cost downstream fine-tuning. Prior work has established these refusal failures, but existing studies do not show how to detect this fragility in the aligned model itself before an attack or fine-tuning inte",
  "authors": "Dongyub Jude Lee, Jungseob Lee, Seungyoon Lee, Seongtae Hong, Suhyune Son, Sugyeong Eo et al.",
  "category": "research",
  "topics": "safety-alignment,healthcare",
  "orgs": null,
  "regions": null,
  "published_at": "2026-06-21T21:30:15.000Z",
  "fetched_at": "2026-07-14T14:14:46.033Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/728",
  "original_url": "https://arxiv.org/abs/2606.22676v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}