{
  "id": 12598,
  "url": "https://arxiv.org/abs/2607.19201v1",
  "title": "MIRA-Ev:A Benchmark for Granular Evidence Detection and Relational Reasoning in Clinical Exams",
  "summary": "Clinical NLP evaluation remains dominated by multiple-choice question answering (MCQA), which scores only final-answer accuracy and cannot detect when a model reaches the correct diagnosis while grounding it in irrelevant, absent, or contradictory evidence. We introduce MIRA-Ev, a clinical argument mining benchmark built on Spanish Médico Interno Residente (MIR) licensing-exam cases, re-annotated by expert clinicians with span-level premises, claims, and directed support/attack relations, and re",
  "authors": "Iker De la Iglesia, Johanna Ramirez-Romero, Jose Maria Villa-Gonzalez, Irune Urroz García, Ander Barrena, Aitziber Atutxa",
  "category": "research",
  "topics": "copyright-ip,healthcare",
  "orgs": null,
  "regions": null,
  "published_at": "2026-07-21T15:34:44.000Z",
  "fetched_at": "2026-07-22T05:10:49.469Z",
  "source_slug": "x-arxiv-cs-ai",
  "source_name": "arXiv cs.AI",
  "source_homepage": "https://arxiv.org/list/cs.AI/recent",
  "ethics_ai_record_url": "https://ethics.ai/record/12598",
  "original_url": "https://arxiv.org/abs/2607.19201v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}