{
  "id": 4696,
  "url": "https://arxiv.org/abs/2605.08496v1",
  "title": "Latent Personality Alignment: Improving Harmlessness Without Mentioning Harms",
  "summary": "Current adversarial robustness methods for large language models require extensive datasets of harmful prompts (thousands to hundreds of thousands of examples), yet remain vulnerable to novel attack vectors and distributional shifts. We propose Latent Personality Alignment (LPA), a sample-efficient defense that achieves robustness by training models on abstract personality traits rather than specific harmful behaviors. Using fewer than 100 trait statements and latent adversarial training, LPA ac",
  "authors": "Linh Le, David Williams-King, Mohamed Amine Merzouk, Aton Kamanda, Adam Oberman",
  "category": "research",
  "topics": "safety-alignment,military-security",
  "orgs": null,
  "regions": null,
  "published_at": "2026-05-08T21:21:59.000Z",
  "fetched_at": "2026-07-14T16:31:12.744Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/4696",
  "original_url": "https://arxiv.org/abs/2605.08496v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}