{
  "id": 5747,
  "url": "https://arxiv.org/abs/2604.15725v1",
  "title": "Reasoning-targeted Jailbreak Attacks on Large Reasoning Models via Semantic Triggers and Psychological Framing",
  "summary": "Large Reasoning Models (LRMs) have demonstrated strong capabilities in generating step-by-step reasoning chains alongside final answers, enabling their deployment in high-stakes domains such as healthcare and education. While prior jailbreak attack studies have focused on the safety of final answers, little attention has been given to the safety of the reasoning process. In this work, we identify a novel problem that injects harmful content into the reasoning steps while preserving unchanged ans",
  "authors": "Zehao Wang, Lanjun Wang",
  "category": "research",
  "topics": "safety-alignment,healthcare,children-education",
  "orgs": null,
  "regions": null,
  "published_at": "2026-04-17T05:56:46.000Z",
  "fetched_at": "2026-07-14T16:31:57.535Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/5747",
  "original_url": "https://arxiv.org/abs/2604.15725v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}