{
  "id": 569,
  "url": "https://arxiv.org/abs/2606.26891v1",
  "title": "Bridging Vision and Language Concepts through Optimal Transport Semantic Flow",
  "summary": "Concept Bottleneck Models (CBMs) promise transparent reasoning by predicting through human-interpretable concepts, yet their effectiveness fundamentally depends on how well visual and textual representations are aligned or matched. Existing vision-language CBMs often rely on pre-aligned encoders or global cosine similarity, which obscures fine-grained concept localization and fails to reflect true semantic geometry. In this work, we rethink concept alignment as a dynamic cross-modal transport pr",
  "authors": "Chenyang Zhang, Anqi Dong, Guangming Zhu, Nuoye Xiong, Siyuan Wang, Lin Mei et al.",
  "category": "research",
  "topics": "safety-alignment,transparency",
  "orgs": null,
  "regions": null,
  "published_at": "2026-06-25T11:24:44.000Z",
  "fetched_at": "2026-07-14T14:14:37.248Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/569",
  "original_url": "https://arxiv.org/abs/2606.26891v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}