{
  "id": 6131,
  "url": "https://arxiv.org/abs/2604.08169v2",
  "title": "Activation Steering for Aligned Open-ended Generation without Sacrificing Coherence",
  "summary": "Alignment in LLMs is more brittle than commonly assumed: misalignment can be induced by adversarial prompts, benign fine-tuning, emergent misalignment, and goal misgeneralization. Recent evidence suggests that some misalignment behaviors are encoded as linear structure in activation space, making it tractable via activation steering, which could be used as a lightweight runtime defense. We implement three methods: Steer-With-Fixed-Coefficient (SwFC), which applies uniform additive steering, and ",
  "authors": "Niklas Herbster, Martin Zborowski, Alberto Tosato, Gauthier Gidel, Tommaso Tosato",
  "category": "research",
  "topics": "safety-alignment,military-security",
  "orgs": null,
  "regions": null,
  "published_at": "2026-04-09T12:28:22.000Z",
  "fetched_at": "2026-07-14T16:32:15.637Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/6131",
  "original_url": "https://arxiv.org/abs/2604.08169v2",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}