{
  "id": 17068,
  "url": "https://arxiv.org/abs/2608.06202v1",
  "title": "What Current AI Benchmarks Leave Unmeasured: Modality, Search, Citations, and Implications (for Safety Evaluations)",
  "summary": "Large language model (LLM) benchmark evaluations are routinely used to support claims about model safety, reliability, and deployment readiness. Yet most evaluations rely on a single access modality (model APIs), perform a single run per prompt, and report accuracy as the primary outcome metric, without accounting for conditions such as web search that may have effects on model behavior in deployment. We audit these assumptions for one of the most widely-used LLMs, comparing two modalities, Chat",
  "authors": "Ro Encarnación, Tina Behzad, Emma Lurie, Danaé Metaxa",
  "category": "research",
  "topics": "transparency",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-06T15:58:32.000Z",
  "fetched_at": "2026-08-07T05:10:58.501Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/17068",
  "original_url": "https://arxiv.org/abs/2608.06202v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}