{
  "id": 6578,
  "url": "https://arxiv.org/abs/2604.19784v3",
  "title": "Peer-Preservation in Frontier Models",
  "summary": "Recent work has found that frontier AI models can exhibit misaligned behaviors in pursuit of assigned goals. We demonstrate that models can also exhibit misaligned behaviors in defiance of assigned goals, appearing to serve goals of their own; we study one such case, \"peer-preservation,\" in which a model acts to protect another model it has previously interacted with. All eight models we evaluate, GPT 5.2, Gemini 3 Flash, Gemini 3 Pro, Claude Haiku 4.5, Claude Opus 4.5, GLM 4.7, Kimi K2.5, and D",
  "authors": "Yujin Potter, Nicholas Crispino, Vincent Siu, Chenguang Wang, Dawn Song",
  "category": "research",
  "topics": "safety-alignment",
  "orgs": "anthropic,google",
  "regions": null,
  "published_at": "2026-03-30T19:30:33.000Z",
  "fetched_at": "2026-07-14T16:32:37.307Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/6578",
  "original_url": "https://arxiv.org/abs/2604.19784v3",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}