{
  "id": 3842,
  "url": "https://arxiv.org/abs/2605.24229v1",
  "title": "How Well Do Models Follow Their Constitutions?",
  "summary": "Frontier AI developers now train models against long written behavioral specifications, such as Anthropic's constitution (Anthropic, 2025a) and OpenAI's Model Spec (OpenAI, 2025a), integrated into post-training via methods like character training (Anthropic, 2024) and deliberative alignment (Guan et al., 2024). These documents serve a governance function, but it is unclear how well models actually follow them under adversarial, multi-turn pressure similar to what they would face in real-world de",
  "authors": "Arya Jakkli, Senthooran Rajamanoharan, Neel Nanda",
  "category": "research",
  "topics": "regulation,safety-alignment",
  "orgs": "openai,anthropic",
  "regions": null,
  "published_at": "2026-05-22T21:17:16.000Z",
  "fetched_at": "2026-07-14T16:30:31.923Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/3842",
  "original_url": "https://arxiv.org/abs/2605.24229v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}