{
  "id": 11968,
  "url": "https://arxiv.org/abs/2607.18114v1",
  "title": "How Does Alignment Tuning Shape Representations of Sycophancy and Related Cue-Induced Biases in LLMs?",
  "summary": "Modern LLMs are alarmingly susceptible to surprisingly simple immaterial changes of input prompts: a casual hint, an incorrectly labeled few-shot example, or a fake prior assistant turn often flips an originally correct answer. We study where this susceptibility, spanning sycophancy and related cue-induced biases, lives inside the model. Across five model families and seven BCT bias types, we extract a per-bias direction from hidden states and triangulate it through three measures: probing, leav",
  "authors": "Prakhar Gupta, Terry Jingchen Zhang, Florent Draye, Bernhard Schölkopf, Zhijing Jin",
  "category": "research",
  "topics": "bias-fairness,safety-alignment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-07-20T16:10:06.000Z",
  "fetched_at": "2026-07-21T05:10:12.656Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/11968",
  "original_url": "https://arxiv.org/abs/2607.18114v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}