{
  "id": 18270,
  "url": "https://arxiv.org/abs/2608.09624v1",
  "title": "Measuring the Wrong Thing: Internal Harmfulness Scores Anti-Rank Successful Jailbreaks",
  "summary": "Internal safety scores judge a prompt before any text is generated, and they are validated by how well they separate harmful prompts from benign ones. That separation is then read as evidence that the score will also catch the attacks that succeed. Harmful intent is a property of the prompt. Jailbreak success is an outcome produced later by a particular target model, decoding policy, and judge. A filter tuned on a score that measures the wrong quantity spends its false positive budget on attacks",
  "authors": "Mingyu Luo, Ming Deng, Zilang Qiu, Yiming Cheng, Ci Tao, Xue Tan, Sijin Sun, Yangfu Li, Ping Chen, Jun Dai, Xiaoyan Sun",
  "category": "research",
  "topics": "regulation,safety-alignment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-10T14:05:32.000Z",
  "fetched_at": "2026-08-11T05:10:37.351Z",
  "source_slug": "x-arxiv-cs-ai",
  "source_name": "arXiv cs.AI",
  "source_homepage": "https://arxiv.org/list/cs.AI/recent",
  "ethics_ai_record_url": "https://ethics.ai/record/18270",
  "original_url": "https://arxiv.org/abs/2608.09624v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}