{
  "id": 17386,
  "url": "https://arxiv.org/abs/2608.05732v1",
  "title": "CircuitSteer: Geometrically Aligned Multi-Layer Steering via Sparse Autoencoder Circuits",
  "summary": "Controlling the behavior of large language models (LLMs) remains a critical challenge for AI alignment. Existing steering methods, such as Contrastive Activation Addition (CAA), typically rely on fixed single-layer interventions derived from aggregate activation differences. These methods impose a single intervention across semantically diverse inputs and often fail to sustain consistent behavioral changes across layers, limiting the effectiveness of the steering. In this work, we introduce Circ",
  "authors": "Mehrshad Saadatinia, Parsa Razmara, Ardalan Aryashad, Ali Abbasi, Seyedarmin Azizi",
  "category": "research",
  "topics": "safety-alignment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-06T08:17:50.000Z",
  "fetched_at": "2026-08-07T05:10:58.501Z",
  "source_slug": "x-arxiv-alignment-query",
  "source_name": "arXiv alignment query",
  "source_homepage": "https://arxiv.org/a/alignment",
  "ethics_ai_record_url": "https://ethics.ai/record/17386",
  "original_url": "https://arxiv.org/abs/2608.05732v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}