{
  "id": 6069,
  "url": "https://arxiv.org/abs/2604.09750v1",
  "title": "Conflicts Make Large Reasoning Models Vulnerable to Attacks",
  "summary": "Large Reasoning Models (LRMs) have achieved remarkable performance across diverse domains, yet their decision-making under conflicting objectives remains insufficiently understood. This work investigates how LRMs respond to harmful queries when confronted with two categories of conflicts: internal conflicts that pit alignment values against each other and dilemmas, which impose mutually contradictory choices, including sacrificial, duress, agent-centered, and social forms. Using over 1,300 promp",
  "authors": "Honghao Liu, Chengjin Xu, Xuhui Jiang, Cehao Yang, Shengming Yin, Zhengwu Ma et al.",
  "category": "research",
  "topics": "safety-alignment,agents-autonomy,finance-investment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-04-10T11:44:57.000Z",
  "fetched_at": "2026-07-14T16:32:15.634Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/6069",
  "original_url": "https://arxiv.org/abs/2604.09750v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}