{
  "id": 4395,
  "url": "https://arxiv.org/abs/2605.13329v1",
  "title": "Tracing Persona Vectors Through LLM Pretraining",
  "summary": "How large language models internally represent high-level behaviors is a core interpretability question with direct relevance to AI safety: it determines what we can detect, audit, or intervene on. Recent work has shown that traits such as evil or sycophancy correspond to linear directions in the internal activations, the so-called persona vectors. Although these vectors are now routinely utilized to inspect and steer model behavior in safety-relevant settings, how these representations are form",
  "authors": "Viktor Moskvoretskii, Dominik Glandorf, Jorge Medina Moreira, Tanja Käser, Robert West",
  "category": "research",
  "topics": "safety-alignment,transparency",
  "orgs": null,
  "regions": null,
  "published_at": "2026-05-13T10:44:23.000Z",
  "fetched_at": "2026-07-14T16:30:59.235Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/4395",
  "original_url": "https://arxiv.org/abs/2605.13329v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}