{
  "id": 17007,
  "url": "https://arxiv.org/abs/2608.05180",
  "title": "The Nuclear Decision-Making Benchmark: Evaluating Frontier LLMs on Nuclear Tendencies",
  "summary": "arXiv:2608.05180v1 Announce Type: new Abstract: The integration of large language models into defense and national-security workflows raises urgent questions about whether frontier models exhibit stable, consistent, and policy-appropriate preferences in high-stakes contexts. We introduce the Nuclear Decision-Making Benchmark (NDM Bench), a targeted evaluation framework of 151 scenarios authored by PhD-credentialed scholars in international relations spanning four domains: escalation (76), arms c",
  "authors": "Benjamin Jensen, Ian Reynolds, Yasir Atalan, Martin Pollack, Austin Woo, Robert Sincero",
  "category": "research",
  "topics": "regulation,military-security",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-07T04:00:00.000Z",
  "fetched_at": "2026-08-07T05:10:58.501Z",
  "source_slug": "arxiv-cscy",
  "source_name": "arXiv cs.CY",
  "source_homepage": "https://arxiv.org/list/cs.CY/recent",
  "ethics_ai_record_url": "https://ethics.ai/record/17007",
  "original_url": "https://arxiv.org/abs/2608.05180",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}