{
  "id": 3391,
  "url": "https://arxiv.org/abs/2606.02630v1",
  "title": "MultiTurnPSB: Evaluating Multi-Turn Jailbreak Attacks an dClassifier-Based Defenses for Medical AI Safety",
  "summary": "Patient-facing medical chatbots are commonly evaluated on single-turn prompts, yet real users push back after refusals, add urgency, and invoke authority. We introduce MultiTurnPSB, a four-turn adversarial extension of PatientSafetyBench, and evaluate GPT-4.1-mini under fixed template, template-adaptive, and live adversarial attacks. Unsafe responses rise from 35% to nearly 80% by Turn 4 under live attack. Under the same adversary, GPT-4.1-mini and Claude Sonnet 4.5 are statistically indistingui",
  "authors": "Anushka Sheoran, Yiduo Hao",
  "category": "research",
  "topics": "safety-alignment,healthcare",
  "orgs": "openai,anthropic",
  "regions": null,
  "published_at": "2026-05-30T10:09:53.000Z",
  "fetched_at": "2026-07-14T16:30:14.368Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/3391",
  "original_url": "https://arxiv.org/abs/2606.02630v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}