{
  "id": 19448,
  "url": "https://arxiv.org/abs/2608.13267v1",
  "title": "How Do VLMs Behave When Blind or Misled? Behavioral Evaluation of VLMs on Scientific Figures",
  "summary": "Existing vision-language model (VLM) benchmarks emphasize perception and reasoning accuracy (how well VLMs describe and reason about what they see in an image), with limited attention to behavioral reliability under uncertainty (how they behave when visual evidence is missing or misleading). We introduce SciFigBench, a diagnostic VLM benchmark for scientific figure understanding that jointly evaluates perception, reasoning, and behavioral reliability under uncertainty. It contains 250 figures wi",
  "authors": "Paul Osemudiame Oamen, Owusu-Banahene Osei, Ananya Mukherjee, Christian Greisinger, Steffen Eger, Pius Onobhayedo, Wei Zhao",
  "category": "research",
  "topics": "healthcare",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-13T14:06:35.000Z",
  "fetched_at": "2026-08-14T05:10:49.168Z",
  "source_slug": "x-arxiv-cs-ai",
  "source_name": "arXiv cs.AI",
  "source_homepage": "https://arxiv.org/list/cs.AI/recent",
  "ethics_ai_record_url": "https://ethics.ai/record/19448",
  "original_url": "https://arxiv.org/abs/2608.13267v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}