{
  "id": 6391,
  "url": "https://arxiv.org/abs/2604.25933v1",
  "title": "A Scoping Review of LLM-as-a-Judge in Healthcare and the MedJUDGE Framework",
  "summary": "As large language models (LLMs) increasingly generate and process clinical text, scalable evaluation has become critical. LLM-as-a-Judge (LaaJ), which uses LLMs to evaluate model outputs, offers a scalable alternative to costly expert review, but its healthcare adoption raises safety and bias concerns. We conducted a PRISMA-ScR scoping review of six databases (January 2020-January 2026), screening 11,727 studies and including 49. The landscape was dominated by evaluation and benchmarking applica",
  "authors": "Chenyu Li, Zohaib Akhtar, Mingu Kwak, Yuelyu Ji, Hang Zhang, Tracey Obi et al.",
  "category": "research",
  "topics": "bias-fairness,healthcare",
  "orgs": null,
  "regions": null,
  "published_at": "2026-04-03T14:50:01.000Z",
  "fetched_at": "2026-07-14T16:32:28.609Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/6391",
  "original_url": "https://arxiv.org/abs/2604.25933v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}