{
  "id": 781,
  "url": "https://arxiv.org/abs/2607.00023v1",
  "title": "Aligning Sentence Embeddings to Human Concepts via Sparse Autoencoders",
  "summary": "Dense sentence embeddings are fundamental to modern Retrieval-Augmented Generation (RAG) systems but suffer from a lack of interpretability due to feature superposition. This opacity hinders the alignment of retrieval processes with human intent, as the entangled representations are difficult to analyze or control. In this work, we propose a method to disentangle the dense representations of sentence transformers (e.g., E5) into human-interpretable concepts using Top-k Sparse Autoencoders (SAEs)",
  "authors": "Wonseok Shin, Songkuk Kim",
  "category": "research",
  "topics": "safety-alignment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-06-19T02:39:50.000Z",
  "fetched_at": "2026-07-14T14:14:46.036Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/781",
  "original_url": "https://arxiv.org/abs/2607.00023v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}