{
  "id": 6479,
  "url": "https://arxiv.org/abs/2604.01473v3",
  "title": "SelfGrader: LLM Jailbreak Detection via Anchored Token-Level Logits",
  "summary": "Large Language Models (LLMs) are powerful tools for answering user queries, yet they remain highly vulnerable to jailbreak attacks. Existing guardrail methods typically rely on internal features or textual responses to detect malicious queries, which either introduce substantial latency or suffer from randomness in text generation. To overcome these limitations, we propose SelfGrader, a lightweight guardrail method that formulates jailbreak detection as a numerical grading problem using anchored",
  "authors": "Zikai Zhang, Rui Hu, Olivera Kotevska, Jiahao Xu",
  "category": "research",
  "topics": "safety-alignment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-04-01T23:29:12.000Z",
  "fetched_at": "2026-07-14T16:32:33.099Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/6479",
  "original_url": "https://arxiv.org/abs/2604.01473v3",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}