{
  "id": 4991,
  "url": "https://arxiv.org/abs/2605.03301v2",
  "title": "SHIELD: A Diverse Clinical Note Dataset and Distilled Small Language Models for Enterprise-Scale De-identification",
  "summary": "De-identification of clinical text is a prerequisite for the secondary use of electronic health records. Existing public benchmarks such as the i2b2 2006 and 2014 corpora are over a decade old and lack the semantic and demographic diversity of modern clinical narratives. Large Language Models (LLMs) reach state-of-the-art zero-shot extraction, but their use at enterprise scale is limited by computational cost and by hospital data governance that restricts sending Protected Health Information (PH",
  "authors": "Jose D. Posada, David Love, Somalee Datta, Priya Desai",
  "category": "research",
  "topics": "regulation,healthcare",
  "orgs": null,
  "regions": null,
  "published_at": "2026-05-05T02:43:55.000Z",
  "fetched_at": "2026-07-14T16:31:26.334Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/4991",
  "original_url": "https://arxiv.org/abs/2605.03301v2",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}