{
  "id": 46,
  "url": "https://arxiv.org/abs/2607.10226v1",
  "title": "When Are Sparse Feature Interventions Actually Localized? Matched Evaluation for SAE-Based Safety Control",
  "summary": "We evaluate when sparse autoencoder (SAE) features act as localized control handles for safety-relevant behavior. This question is difficult because apparent success can arise from weak interventions, mismatched baselines, model robustness, or degenerate outputs that automated safety judges mark as unsafe without representing meaningful harmful compliance. We introduce a matched coherence-gated evaluation protocol for runtime safety interventions: methods are compared at matched target-effect po",
  "authors": "Daming Luo",
  "category": "research",
  "topics": "regulation",
  "orgs": null,
  "regions": null,
  "published_at": "2026-07-11T09:31:45.000Z",
  "fetched_at": "2026-07-14T14:14:15.665Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/46",
  "original_url": "https://arxiv.org/abs/2607.10226v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}