{
  "id": 4390,
  "url": "https://arxiv.org/abs/2605.13530v1",
  "title": "Towards Unified Surgical Scene Understanding:Bridging Reasoning and Grounding via MLLMs",
  "summary": "Surgical scene understanding is a cornerstone of computer-assisted intervention. While recent advances, particularly in surgical image segmentation, have driven progress, real-world clinical applications require a more holistic understanding that jointly captures procedural context, semantic reasoning, and precise visual grounding. However, existing approaches typically address these components in isolation, leading to fragmented representations and limited semantic consistency. To address this ",
  "authors": "Jincai Huang, Shihao Zou, Yuchen Guo, Jingjing Li, Wei Ji, Kai Wang et al.",
  "category": "research",
  "topics": "healthcare",
  "orgs": null,
  "regions": null,
  "published_at": "2026-05-13T13:42:23.000Z",
  "fetched_at": "2026-07-14T16:30:59.235Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/4390",
  "original_url": "https://arxiv.org/abs/2605.13530v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}