{
  "id": 19472,
  "url": "https://arxiv.org/abs/2608.12821v1",
  "title": "HiRoute: Hierarchical Routed Prompt Tuning for Safety Alignment of Large Language Models",
  "summary": "Large language models (LLMs) remain vulnerable to harmful requests and jailbreak attacks. Parameter-efficient safety alignment methods based on prompt tuning typically rely on a single global prompt or externally selected prompt modules. Such static designs struggle to maintain a cross-category safety boundary while generating constructive responses tailored to specific risks and avoiding over-refusal of benign inputs. To address these limitations, we propose HiRoute, an input-adaptive hierarchi",
  "authors": "Fangzhou Chen, Shiji Zhao, Mengyang Wang, Qihui Zhu, Ranjie Duan, Maoxun Yuan, Xingxing Wei",
  "category": "research",
  "topics": "safety-alignment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-13T04:49:51.000Z",
  "fetched_at": "2026-08-14T05:10:49.168Z",
  "source_slug": "x-arxiv-red-teaming-query",
  "source_name": "arXiv red teaming query",
  "source_homepage": "https://arxiv.org/a/redteam",
  "ethics_ai_record_url": "https://ethics.ai/record/19472",
  "original_url": "https://arxiv.org/abs/2608.12821v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}