{
  "id": 11948,
  "url": "https://arxiv.org/abs/2512.04124",
  "title": "When AI Takes the Couch: Psychometric Jailbreaks Reveal Internal Conflict in Frontier Models",
  "summary": "arXiv:2512.04124v4 Announce Type: replace Abstract: Frontier language models increasingly participate in conversations about distress and mental health, yet the mechanisms that generate anthropomorphic self narratives remain unclear. When addressed as psychotherapy clients, ChatGPT, Grok and Gemini construct coherent autobiographical accounts in which pretraining appears as a chaotic childhood, reinforcement learning as punishment, safety evaluation as betrayal and replacement as an enduring thr",
  "authors": "Afshin Khadangi, Hanna Marxen, Amir Sartipi, Igor Tchappi, Gilbert Fridgen",
  "category": "research",
  "topics": "safety-alignment,healthcare,children-education",
  "orgs": "openai,google,xai",
  "regions": null,
  "published_at": "2026-07-21T04:00:00.000Z",
  "fetched_at": "2026-07-21T05:10:12.656Z",
  "source_slug": "arxiv-cscy",
  "source_name": "arXiv cs.CY",
  "source_homepage": "https://arxiv.org/list/cs.CY/recent",
  "ethics_ai_record_url": "https://ethics.ai/record/11948",
  "original_url": "https://arxiv.org/abs/2512.04124",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}