{
  "id": 17506,
  "url": "https://www.jmir.org/2026/1/e92183",
  "title": "A Bilingual Benchmark for Evaluating Diagnostic Performance of Multimodal Large Language Models in Radiology (RadM-Bench): Evaluation Development and Validation",
  "summary": "Background: Multimodal large language models are increasingly used in radiological diagnosis, but their performance has not been systematically evaluated across volumetric (3D) imaging, real-world clinical versus public teaching cases, and bilingual contexts. Objective: The aim of the study is to develop a bilingual radiology benchmark and characterize the diagnostic performance of state-of-the-art multimodal large language models across input modality, clinical setting (public teaching vs routi",
  "authors": "Qingxia Wu, Qingxia Wu, Peipei Zhang, Zhifeng Yi, Yu Shen, Yan Bai, Hongna Tan, Pei Dong, Zhong Xue, Neil Roberts, Meiyun Wang",
  "category": "research",
  "topics": "healthcare",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-07T20:15:11.000Z",
  "fetched_at": "2026-08-08T05:10:34.355Z",
  "source_slug": "x-jmir-journal-of-medical-internet-researc",
  "source_name": "JMIR (Journal of Medical Internet Research)",
  "source_homepage": "https://www.jmir.org",
  "ethics_ai_record_url": "https://ethics.ai/record/17506",
  "original_url": "https://www.jmir.org/2026/1/e92183",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}