{
  "id": 16570,
  "url": "https://arxiv.org/abs/2608.02674v1",
  "title": "Moving the Safety Barrier: Dynamic Routing Adaptive Alignment Against White-Box Attacks",
  "summary": "With the widespread deployment of large foundation models (LFMs) in open environments, safety threats are shifting from black-box jailbreaks toward white-box attacks that directly identify and disrupt internal safety neurons or routes. However, existing safety defenses often rely on static safety units or fixed refusal pathways, leaving models highly vulnerable to targeted route-level white-box attacks. For that, we propose dynamic routing adaptive alignment (DRAA), a framework that introduces d",
  "authors": "Shangze Li, Chuancheng Shi, Simiao Xie, Lingzhi He, Cheng Ji, Zifeng Cheng, Fei Shen, Chao Wu, Tat-Seng Chua",
  "category": "research",
  "topics": "safety-alignment,environment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-02T17:45:20.000Z",
  "fetched_at": "2026-08-05T05:10:44.550Z",
  "source_slug": "x-arxiv-red-teaming-query",
  "source_name": "arXiv red teaming query",
  "source_homepage": "https://arxiv.org/a/redteam",
  "ethics_ai_record_url": "https://ethics.ai/record/16570",
  "original_url": "https://arxiv.org/abs/2608.02674v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}