{
  "id": 5813,
  "url": "https://arxiv.org/abs/2604.14306v2",
  "title": "EuropeMedQA Study Protocol: A Multilingual, Multimodal Medical Examination Dataset for Language Model Evaluation",
  "summary": "While Large Language Models (LLMs) have demonstrated high proficiency on English-centric medical examinations, their performance often declines when faced with non-English languages and multimodal diagnostic tasks. This study protocol describes the development of EuropeMedQA, the first comprehensive, multilingual, and multimodal medical examination dataset sourced from official regulatory exams in Italy, France, Spain, and Portugal. Following FAIR data principles and SPIRIT-AI guidelines, we des",
  "authors": "Francesco Andrea Causio, Vittorio De Vita, Olivia Riccomi, Michele Ferramola, Federico Felizzi, Alessandro Tosi et al.",
  "category": "research",
  "topics": "regulation,healthcare",
  "orgs": null,
  "regions": "eu",
  "published_at": "2026-04-15T18:03:13.000Z",
  "fetched_at": "2026-07-14T16:32:02.060Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/5813",
  "original_url": "https://arxiv.org/abs/2604.14306v2",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}