{
  "id": 5101,
  "url": "https://arxiv.org/abs/2607.01239v1",
  "title": "Breaking Safety at the Token Boundary: How BPE Tokenization Creates Exploitable Gaps in LLM Alignment",
  "summary": "Character-level perturbations bypass safety alignment in modern LLMs despite leaving prompts human-readable. We identify and test a central structural mechanism: BPE tokenization fragments safety-critical words into sub-word pieces, and the three public alignment datasets we surveyed contain no intentionally fragmented inputs. The mechanism is a chain, tested end-to-end on five model families (Qwen-3-4B, Qwen-2.5-7B, Gemma-3-4B, Llama-3.1-8B, Mistral-7B). An optimization targeting safety-token f",
  "authors": "Tung-Ling Li, Hongliang Liu, Yuhao Wu",
  "category": "research",
  "topics": "safety-alignment",
  "orgs": "mistral",
  "regions": null,
  "published_at": "2026-05-01T18:57:03.000Z",
  "fetched_at": "2026-07-14T16:31:31.210Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/5101",
  "original_url": "https://arxiv.org/abs/2607.01239v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}