{
  "id": 6163,
  "url": "https://arxiv.org/abs/2604.07729v1",
  "title": "Emotion Concepts and their Function in a Large Language Model",
  "summary": "Large language models (LLMs) sometimes appear to exhibit emotional reactions. We investigate why this is the case in Claude Sonnet 4.5 and explore implications for alignment-relevant behavior. We find internal representations of emotion concepts, which encode the broad concept of a particular emotion and generalize across contexts and behaviors it might be linked to. These representations track the operative emotion concept at a given token position in a conversation, activating in accordance wi",
  "authors": "Nicholas Sofroniew, Isaac Kauvar, William Saunders, Runjin Chen, Tom Henighan, Sasha Hydrie et al.",
  "category": "research",
  "topics": "safety-alignment,finance-investment",
  "orgs": "anthropic",
  "regions": null,
  "published_at": "2026-04-09T02:25:17.000Z",
  "fetched_at": "2026-07-14T16:32:20.052Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/6163",
  "original_url": "https://arxiv.org/abs/2604.07729v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}