{
  "id": 6497,
  "url": "https://arxiv.org/abs/2604.00788v1",
  "title": "UK AISI Alignment Evaluation Case-Study",
  "summary": "This technical report presents methods developed by the UK AI Security Institute for assessing whether advanced AI systems reliably follow intended goals. Specifically, we evaluate whether frontier models sabotage safety research when deployed as coding assistants within an AI lab. Applying our methods to four frontier models, we find no confirmed instances of research sabotage. However, we observe that Claude Opus 4.5 Preview (a pre-release snapshot of Opus 4.5) and Sonnet 4.5 frequently refuse",
  "authors": "Alexandra Souly, Robert Kirk, Jacob Merizian, Abby D'Cruz, Xander Davies",
  "category": "research",
  "topics": "safety-alignment",
  "orgs": "anthropic",
  "regions": "uk",
  "published_at": "2026-04-01T11:53:25.000Z",
  "fetched_at": "2026-07-14T16:32:33.100Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/6497",
  "original_url": "https://arxiv.org/abs/2604.00788v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}