{
  "id": 7520,
  "url": "https://arxiv.org/abs/2603.13359v1",
  "title": "From Refusal Tokens to Refusal Control: Discovering and Steering Category-Specific Refusal Directions",
  "summary": "Language models are commonly fine-tuned for safety alignment to refuse harmful prompts. One approach fine-tunes them to generate categorical refusal tokens that distinguish different refusal types before responding. In this work, we leverage a version of Llama 3 8B fine-tuned with these categorical refusal tokens to enable inference-time control over fine-grained refusal behavior, improving both safety and reliability. We show that refusal token fine-tuning induces separable, category-aligned di",
  "authors": "Rishab Alagharu, Ishneet Sukhvinder Singh, Shaibi Shamsudeen, Zhen Wu, Ashwinee Panda",
  "category": "research",
  "topics": "safety-alignment",
  "orgs": "meta",
  "regions": null,
  "published_at": "2026-03-09T06:37:16.000Z",
  "fetched_at": "2026-07-14T16:33:16.669Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/7520",
  "original_url": "https://arxiv.org/abs/2603.13359v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}