{
  "id": 4758,
  "url": "https://arxiv.org/abs/2605.07324v1",
  "title": "Activation Differences Reveal Backdoors: A Comparison of SAE Architectures",
  "summary": "Backdoor attacks on language models pose a significant threat to AI safety, where models behave normally on most inputs but exhibit harmful behavior when triggered by specific patterns. Detecting such backdoors through mechanistic interpretability remains an open challenge. We investigate two sparse autoencoder architectures -- Crosscoders and Differential SAEs (Diff-SAE) -- for isolating backdoor-related features in fine-tuned models. Using a controlled SQL injection backdoor triggered by year-",
  "authors": "Sachin Kumar",
  "category": "research",
  "topics": "safety-alignment,finance-investment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-05-08T06:30:26.000Z",
  "fetched_at": "2026-07-14T16:31:12.748Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/4758",
  "original_url": "https://arxiv.org/abs/2605.07324v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}