{
  "id": 7579,
  "url": "https://arxiv.org/abs/2603.06816v1",
  "title": "\"Dark Triad\" Model Organisms of Misalignment: Narrow Fine-Tuning Mirrors Human Antisocial Behavior",
  "summary": "The alignment problem refers to concerns regarding powerful intelligences, ensuring compatibility with human preferences and values as capabilities increase. Current large language models (LLMs) show misaligned behaviors, such as strategic deception, manipulation, and reward-seeking, that can arise despite safety training. Gaining a mechanistic understanding of these failures requires empirical approaches that can isolate behavioral patterns in controlled settings. We propose that biological mis",
  "authors": "Roshni Lulla, Fiona Collins, Sanaya Parekh, Thilo Hagendorff, Jonas Kaplan",
  "category": "research",
  "topics": "safety-alignment,biotech",
  "orgs": null,
  "regions": null,
  "published_at": "2026-03-06T19:23:21.000Z",
  "fetched_at": "2026-07-14T16:33:21.047Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/7579",
  "original_url": "https://arxiv.org/abs/2603.06816v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}