{
  "id": 4059,
  "url": "https://arxiv.org/abs/2605.19321v1",
  "title": "Exploring and Developing a Pre-Model Safeguard with Draft Models",
  "summary": "Large Language Model (LLM) alignment remains vulnerable to jailbreak attacks that elicit unsafe responses, motivating pre-model and post-model guards. Pre-model guards audit the safety of prompts before invoking target models. However, relying solely on the prompt often leads to high false-negative rates (i.e., jailbreak attacks go undetected). Post-model guards address this issue by auditing both the user prompt and the target model's response. However, they incur a high computational cost, inc",
  "authors": "Hongyu Cai, Arjun Arunasalam, Yiming Liang, Antonio Bianchi, Z. Berkay Celik",
  "category": "research",
  "topics": "safety-alignment,transparency",
  "orgs": null,
  "regions": null,
  "published_at": "2026-05-19T04:01:36.000Z",
  "fetched_at": "2026-07-14T16:30:41.584Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/4059",
  "original_url": "https://arxiv.org/abs/2605.19321v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}