{
  "id": 6123,
  "url": "https://arxiv.org/abs/2605.22826v1",
  "title": "Evaluating Large Language Models in a Complex Hidden Role Game",
  "summary": "Quantifying the deceptive potential of Large Language Models (LLMs) is critical for AI safety, yet difficult to achieve in uncontrolled environments. This work investigates the reasoning, persuasion, and deceptive capabilities of LLMs within the social deduction game Secret Hitler. I introduce an open-source framework and novel metrics to measure performance: Role Identification Accuracy, Deception Retention Rate, and Game State Impact Rate. By benchmarking models against rule-based algorithms a",
  "authors": "Niklas Bauer",
  "category": "research",
  "topics": "safety-alignment,environment,finance-investment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-04-09T14:02:14.000Z",
  "fetched_at": "2026-07-14T16:32:15.636Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/6123",
  "original_url": "https://arxiv.org/abs/2605.22826v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}