{
  "id": 18421,
  "url": "https://arxiv.org/abs/2608.10635v1",
  "title": "MedUP: Awakening Unified Understanding and Perception in Medical Vision-Language Models",
  "summary": "Medical Vision-Language Models (Med-VLMs) excel at verbalizing visual content, yet precise visual perception, segmentation, and grounding remain challenging. Existing approaches either verbalize regions as coordinate strings or rely on external modules that decouple perception from understanding, creating representation gaps for region-language alignment. We present MedUP, a Med-VLM that natively unifies perception and understanding within a shared token space. At its core lies UniMedTok, a regi",
  "authors": "Yuan Wang, Hualiang Wang, Yixin Chen, Songtao Jiang, Shujian Gao, Jiaming Lin et al.",
  "category": "research",
  "topics": "safety-alignment,healthcare",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-11T08:22:13.000Z",
  "fetched_at": "2026-08-12T05:10:43.828Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/18421",
  "original_url": "https://arxiv.org/abs/2608.10635v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}