{
  "id": 5047,
  "url": "https://arxiv.org/abs/2605.01899v1",
  "title": "Disentangling Intent from Role: Adversarial Self-Play for Persona-Invariant Safety Alignment",
  "summary": "The growing capabilities of large language models (LLMs) have driven their widespread deployment across diverse domains, even in potentially high-risk scenarios. Despite advances in safety alignment techniques, current models remain vulnerable to emerging persona-based jailbreak attacks. Existing research on persona-based jailbreak has primarily focused on attack iterations, yet it lacks systemic and mechanistic constraints on the defense side. To address this challenge, we propose Persona-Invar",
  "authors": "Jiajia Li, Xiaoyu Wen, Zhongtian Ma, Shuyue Hu, Qiaosheng Zhang, Zhen Wang",
  "category": "research",
  "topics": "safety-alignment,military-security",
  "orgs": null,
  "regions": null,
  "published_at": "2026-05-03T14:28:08.000Z",
  "fetched_at": "2026-07-14T16:31:26.338Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/5047",
  "original_url": "https://arxiv.org/abs/2605.01899v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}