{
  "id": 18409,
  "url": "https://arxiv.org/abs/2608.10954v1",
  "title": "Evidence-Grounded Trustworthy Multimodal Reasoning and Evaluation Benchmark in Complex Urban Scenes",
  "summary": "While Multimodal Large Language Models (MLLMs) demonstrate impressive performance in benign scenarios, their cognitive reliability deteriorates significantly in complex scenes under adverse conditions. In these settings, models often rely on implicit inference without sufficient visual evidence, leading to a disconnect between perception and reasoning. Meanwhile, existing outcome-oriented benchmarks evaluate only final predictions and fail to diagnose failures in the underlying reasoning process",
  "authors": "Zhaoyang Wei, Bowen Jiang, Xumeng Han, Jiashu Li, Xuehui Yu, Yuling Liu et al.",
  "category": "research",
  "topics": "healthcare",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-11T14:23:07.000Z",
  "fetched_at": "2026-08-12T05:10:43.828Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/18409",
  "original_url": "https://arxiv.org/abs/2608.10954v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}