{
  "id": 4642,
  "url": "https://arxiv.org/abs/2605.11002v1",
  "title": "MT-JailBench: A Modular Benchmark for Understanding Multi-Turn Jailbreak Attacks",
  "summary": "Multi-turn jailbreaks exploit the ability of large language models to accumulate and act on conversational context. Instead of stating a harmful request directly, an attacker can gradually steer the conversation toward an unsafe answer. Recent methods demonstrate this risk, but they are usually evaluated as black-box pipelines with different budgets, judges, retry rules, and strategy generation procedures. As a result, it is often unclear whether reported gains reflect stronger attack mechanisms",
  "authors": "Xinkai Zhang, Zhipeng Wei, Huanli Gong, Jing Ting Zheng, Yuchen Zhang, Yue Dong et al.",
  "category": "research",
  "topics": "safety-alignment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-05-10T00:17:14.000Z",
  "fetched_at": "2026-07-14T16:31:08.357Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/4642",
  "original_url": "https://arxiv.org/abs/2605.11002v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}