{
  "id": 18693,
  "url": "https://arxiv.org/abs/2608.11167v1",
  "title": "MultiModal Code-Switching: Interleaving Visual Objects into Language for Explicit Object-Level Alignment",
  "summary": "Existing Multimodal Large Language Models (MLLMs) predominantly rely on image-text pairs for modality alignment pretraining, mapping global image representations to long textual descriptions. However, this image-level alignment suffers from referential ambiguity: models struggle to infer the correspondences between multiple visual objects and textual entities from the global representation, leading to data inefficiency and suboptimal semantic grounding. To address this, we propose MultiModal Cod",
  "authors": "Changhao Xiang, Shangyu Xing, Zhen Wu, Jianbing Zhang, Xinyu Dai",
  "category": "research",
  "topics": "safety-alignment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-11T17:28:52.000Z",
  "fetched_at": "2026-08-12T05:10:43.828Z",
  "source_slug": "arxiv-cslg",
  "source_name": "arXiv cs.LG",
  "source_homepage": "https://arxiv.org/list/cs.LG/recent",
  "ethics_ai_record_url": "https://ethics.ai/record/18693",
  "original_url": "https://arxiv.org/abs/2608.11167v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}