{
  "id": 14569,
  "url": "https://arxiv.org/abs/2607.27081v1",
  "title": "On-Policy Distillation for LLM Safety: A Routing Approach to Template-Robust Realignment",
  "summary": "Fine-tuning is the dominant paradigm for specializing large language models (LLMs), yet it exposes a critical vulnerability: malicious data providers can embed harmful behaviors into downstream corpora, creating models that retain professional skills while violating human values on demand. Existing safety-realignment defenses often fail in practice due to three key limitations: they frequently cause catastrophic forgetting of specialized skills; their effectiveness collapses when the defender ca",
  "authors": "Yongjian Guo, Wanlun Ma, Lingyu Shen, Xi Xiao, Sheng Wen",
  "category": "research",
  "topics": "regulation",
  "orgs": null,
  "regions": null,
  "published_at": "2026-07-29T16:07:19.000Z",
  "fetched_at": "2026-07-30T05:10:24.387Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/14569",
  "original_url": "https://arxiv.org/abs/2607.27081v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}