{
  "id": 6424,
  "url": "https://arxiv.org/abs/2604.02574v1",
  "title": "Understanding the Effects of Safety Unalignment on Large Language Models",
  "summary": "Safety alignment has become a critical step to ensure LLMs refuse harmful requests while providing helpful and harmless responses. However, despite the ubiquity of safety alignment for deployed frontier models, two separate lines of recent work--jailbreak-tuning (JT) and weight orthogonalization (WO)--have shown that safety guardrails may be largely disabled, resulting in LLMs which comply with harmful requests they would normally refuse. In spite of far-reaching safety implications, analysis ha",
  "authors": "John T. Halloran",
  "category": "research",
  "topics": "safety-alignment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-04-02T23:09:43.000Z",
  "fetched_at": "2026-07-14T16:32:28.611Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/6424",
  "original_url": "https://arxiv.org/abs/2604.02574v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}