{
  "id": 460,
  "url": "https://arxiv.org/abs/2606.29586v1",
  "title": "SonoCLIP: Mask-Guided Region-Aware Vision-Language Pretraining for Fetal Ultrasound Analysis",
  "summary": "Vision-language foundation models have shown strong potential in medical image analysis. Although foundation models for ultrasound imaging have recently emerged, the domain remains particularly challenging due to severe speckle noise, acquisition variability, and subtle anatomical boundaries, leading to high inter-observer variability. Existing CLIP-based models rely primarily on global image-text alignment, limiting their sensitivity to clinically decisive local structures. We propose SonoCLIP,",
  "authors": "Hang Su, Chao Sun, Zhaofan Li, Wei Hu, Juhua Liu, Bo Du",
  "category": "research",
  "topics": "safety-alignment,healthcare",
  "orgs": null,
  "regions": null,
  "published_at": "2026-06-28T20:04:49.000Z",
  "fetched_at": "2026-07-14T14:14:32.648Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/460",
  "original_url": "https://arxiv.org/abs/2606.29586v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}