{
  "id": 14527,
  "url": "https://arxiv.org/abs/2503.10647",
  "title": "The Reliability of LLMs for Medical Diagnosis: An Examination of Consistency, Manipulation, and Contextual Awareness",
  "summary": "arXiv:2503.10647v2 Announce Type: replace-cross Abstract: This study evaluated the diagnostic reliability of two Large Language Models (LLMs), Google Gemini 2.0 Flash and OpenAI ChatGPT-4o, across three dimensions: consistency under rephrased inputs, susceptibility to irrelevant prompt content, and responsiveness to added clinical context. We designed 52 clinical scenarios and modified each under controlled conditions. For consistency, scenarios were rephrased with demographic, wording, and exam",
  "authors": "Krishna Subedi",
  "category": "research",
  "topics": "healthcare",
  "orgs": "openai,google",
  "regions": null,
  "published_at": "2026-07-30T04:00:00.000Z",
  "fetched_at": "2026-07-30T05:10:24.387Z",
  "source_slug": "arxiv-cscy",
  "source_name": "arXiv cs.CY",
  "source_homepage": "https://arxiv.org/list/cs.CY/recent",
  "ethics_ai_record_url": "https://ethics.ai/record/14527",
  "original_url": "https://arxiv.org/abs/2503.10647",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}