{
  "id": 4599,
  "url": "https://arxiv.org/abs/2605.09893v2",
  "title": "Pseudo-Deliberation in Language Models: When Reasoning Fails to Align Values and Actions",
  "summary": "Large language models (LLMs) are often evaluated based on their stated values, yet these do not reliably translate into their actions, a discrepancy termed \"value-action gap.\" In this work, we argue that this gap persists even under explicit reasoning, revealing a deeper failure mode we call \"Pseudo-Deliberation\": the appearance of principled reasoning without corresponding behavioral alignment. To study this systematically, we introduce VALDI, a framework for measuring alignment between stated ",
  "authors": "Sushrita Rakshit, Hanwen Zhang, Hua Shen",
  "category": "research",
  "topics": "safety-alignment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-05-11T02:32:53.000Z",
  "fetched_at": "2026-07-14T16:31:08.354Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/4599",
  "original_url": "https://arxiv.org/abs/2605.09893v2",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}