{
  "id": 1050,
  "url": "https://arxiv.org/abs/2606.20676v1",
  "title": "Jury Duty: Calibration and Orientation Failures in MLLM-as-a-Judge Under Cultural Ambiguity",
  "summary": "MLLM-as-a-Judge is conventionally validated by agreement with human annotations, but this metric is undefined when the human pool is culturally heterogeneous. We introduce VOIR DIRE, a multimodal benchmark of 626 culturally paired image--prompt artifacts spanning U.S. and mainland Chinese contexts across food, fashion, and architecture, with annotator pools that are within-pool reliable (a = 0.86/0.74) but cross-pool divergent on evaluation (Q1 r = -0.12). Across six MLLMs, the bias decomposes i",
  "authors": "Daniel Lee, Harsh Sharma, Eunkyu Park, Pranav Narayanan Venkit, Jeonghwan Kim, Kah Mun Chia et al.",
  "category": "research",
  "topics": "bias-fairness",
  "orgs": null,
  "regions": "china",
  "published_at": "2026-06-12T15:53:40.000Z",
  "fetched_at": "2026-07-14T14:14:59.014Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/1050",
  "original_url": "https://arxiv.org/abs/2606.20676v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}