{
  "id": 16667,
  "url": "https://arxiv.org/abs/2608.04472v1",
  "title": "EndoVLM: An Endoscopy Vision-Language Pre-training Model via Anatomy-Guided Sparsity and Progressive Alignment",
  "summary": "The development of foundation models (FMs) is crucial for advancing endoscopic image analysis. However, existing endoscopy FMs mainly rely on self-supervised learning from uni-modal images or videos, overlooking the rich semantic knowledge contained in clinical reports. Furthermore, effectively leveraging these records is hindered by a fundamental modality gap: structured anatomical descriptions are not naturally mapped to specific frames within the high-redundancy, uncurated visual streams. In",
  "authors": "Zhenyu Yi, Jianwei Xu, Yue Hu, Zhongwei Qiu, Sijing Li, Liang Huang et al.",
  "category": "research",
  "topics": "safety-alignment,healthcare",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-05T05:56:41.000Z",
  "fetched_at": "2026-08-06T05:10:11.148Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/16667",
  "original_url": "https://arxiv.org/abs/2608.04472v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}