{
  "id": 18313,
  "url": "https://arxiv.org/abs/2608.08542v1",
  "title": "When Skills Meet Safety: Benchmarking and Characterizing the Adaptive Jailbreak Robustness of Skill-Merged LLMs",
  "summary": "Model merging has become the default way to give an aligned language model new skills without retraining: a practitioner folds task vectors from math, code, or domain specialists into a safety-aligned base using task arithmetic, TIES, or DARE. This convenience is known to carry a safety cost, but almost all of that evidence rests on static refusal tests: fixed harmful prompts scored for compliance. We argue this is misleading. Because safety alignment is \"shallow,\" concentrated in the first few",
  "authors": "Yu Ma, Hongli Shi, Jing Li, Xinran Xu, Weiwei Hou",
  "category": "research",
  "topics": "regulation,safety-alignment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-09T07:41:06.000Z",
  "fetched_at": "2026-08-11T05:10:37.351Z",
  "source_slug": "x-arxiv-red-teaming-query",
  "source_name": "arXiv red teaming query",
  "source_homepage": "https://arxiv.org/a/redteam",
  "ethics_ai_record_url": "https://ethics.ai/record/18313",
  "original_url": "https://arxiv.org/abs/2608.08542v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}