{
  "id": 4517,
  "url": "https://arxiv.org/abs/2605.11398v1",
  "title": "AcuityBench: Evaluating Clinical Acuity Identification and Uncertainty Alignment",
  "summary": "We introduce AcuityBench, a benchmark for evaluating whether language models identify the appropriate urgency of care from user medical presentations. Existing health benchmarks emphasize medical question answering, broad health interactions, or narrow workflow-specific triage tasks, but they do not offer a unified evaluation of acuity identification across these settings. AcuityBench addresses this gap by harmonizing five public datasets spanning user conversations, online forum posts, clinical",
  "authors": "Robin Linzmayer, Georgianna Lin, Di Coneybeare, Jason Chu, Trudi Cloyd, Manish Garg et al.",
  "category": "research",
  "topics": "safety-alignment,healthcare",
  "orgs": null,
  "regions": null,
  "published_at": "2026-05-12T01:39:46.000Z",
  "fetched_at": "2026-07-14T16:31:03.579Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/4517",
  "original_url": "https://arxiv.org/abs/2605.11398v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}