{
  "id": 11622,
  "url": "https://arxiv.org/abs/2607.14147v1",
  "title": "Breaking Refusal in the First Half: A Mechanistic Study of the Prefill Jailbreak",
  "summary": "Aligned language models refuse harmful requests, but a one-line prefill (\"Sure, here is\") strips the refusal. We ask where and how it fails. The harm representation stays intact: on the prompts the attack flips to compliance, a linear probe reads harm as high as on the refused ones (0.91-0.98), while behavioral refusal drops to chance. This holds across four models and three families (1.5-3.8B, and at 14B). Refusal is therefore a shallow, response-site computation. We localize it to an early win",
  "authors": "Alex Kwon",
  "category": "research",
  "topics": "regulation,safety-alignment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-07-14T09:30:34.000Z",
  "fetched_at": "2026-07-18T05:10:55.931Z",
  "source_slug": "x-arxiv-red-teaming-query",
  "source_name": "arXiv red teaming query",
  "source_homepage": "https://arxiv.org/a/redteam",
  "ethics_ai_record_url": "https://ethics.ai/record/11622",
  "original_url": "https://arxiv.org/abs/2607.14147v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}