{
  "id": 4129,
  "url": "https://arxiv.org/abs/2605.18194v1",
  "title": "Beyond the Cartesian Illusion: Testing Two-Stage Multi-Modal Theory of Mind under Perceptual Bottlenecks",
  "summary": "While Multi-Modal Large Language Models (MLLMs) demonstrate impressive capabilities in general reasoning, their embodied spatial intelligence remains hampered by a \"Cartesian Illusion\" - a reliance on text-based probability distributions that lack grounded, 3D topological understanding. This limitation is starkly exposed in multi-agent environments, which demand more than just scene perception; they require second-order Theory of Mind (ToM). Specifically, an Agent A must be able to infer Agent B",
  "authors": "Yajing Zhou, Xiangyu Kong",
  "category": "research",
  "topics": "agents-autonomy,environment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-05-18T10:32:56.000Z",
  "fetched_at": "2026-07-14T16:30:45.940Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/4129",
  "original_url": "https://arxiv.org/abs/2605.18194v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}