{
  "id": 6380,
  "url": "https://arxiv.org/abs/2604.03436v1",
  "title": "MetaSAEs: Joint Training with a Decomposability Penalty Produces More Atomic Sparse Autoencoder Latents",
  "summary": "Sparse autoencoders (SAEs) are increasingly used for safety-relevant applications including alignment detection and model steering. These use cases require SAE latents to be as atomic as possible. Each latent should represent a single coherent concept drawn from a single underlying representational subspace. In practice, SAE latents blend representational subspaces together. A single feature can activate across semantically distinct contexts that share no true common representation, muddying an ",
  "authors": "Matthew Levinson",
  "category": "research",
  "topics": "safety-alignment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-04-03T20:20:28.000Z",
  "fetched_at": "2026-07-14T16:32:28.608Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/6380",
  "original_url": "https://arxiv.org/abs/2604.03436v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}