{
  "id": 9949,
  "url": "https://doi.org/10.1038/s41746-026-02428-5",
  "title": "Large language models provide unsafe answers to patient-posed medical questions",
  "summary": "Millions of patients are regularly using large language model (LLM) chatbots for medical advice, raising patient safety concerns. This physician-led red-teaming study compares the safety of four publicly available chatbots-Claude by Anthropic, Gemini by Google, GPT-4o by OpenAI, and Llama-3.0/3.1-70B by Meta-on a new dataset, HealthAdvice, using an evaluation framework that enables quantitative and qualitative analysis. In total, 888 chatbot responses are evaluated for 222 patient-posed advice-s",
  "authors": "Rachel Lea Draelos, Samina Afreen, Barbara Blasko, Tiffany L. Brazile, Natasha Chase, Dimple Patel Desai",
  "category": "research",
  "topics": "safety-alignment,healthcare",
  "orgs": "openai,anthropic,google,meta",
  "regions": null,
  "published_at": "2026-02-13T00:00:00.000Z",
  "fetched_at": "2026-07-14T16:34:04.099Z",
  "source_slug": "openalex",
  "source_name": "OpenAlex",
  "source_homepage": "https://openalex.org",
  "ethics_ai_record_url": "https://ethics.ai/record/9949",
  "original_url": "https://doi.org/10.1038/s41746-026-02428-5",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}