{
  "id": 18787,
  "url": "https://arxiv.org/abs/2608.11806v1",
  "title": "Understanding Content Moderation in Large Language Models through Restricted Books: From Refusal to Warning",
  "summary": "As large language models enter everyday information pipelines, understanding how they handle sensitive topics matters as much as understanding whether they handle them at all. We study this question through a large-scale, systematic experiment using restricted versus unrestricted books as a controlled testbed: 40,800 query-response pairs, 400 books, 17 prompt designs, and six frontier models spanning six AI providers (Claude Sonnet 4.5, GPT-4o, Gemini 2.5 Flash, DeepSeek-V3, Qwen-Plus, and Grok-",
  "authors": "Xucheng Yu, Emily Knox, Haohan Wang",
  "category": "research",
  "topics": null,
  "orgs": "anthropic,google,xai,deepseek",
  "regions": null,
  "published_at": "2026-08-12T08:51:13.000Z",
  "fetched_at": "2026-08-13T05:10:37.786Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/18787",
  "original_url": "https://arxiv.org/abs/2608.11806v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}