{
  "id": 15692,
  "url": "https://arxiv.org/abs/2607.28801v1",
  "title": "Benchmarks Are Not Monolithic: Sample-Level Auditing and Orchestration for LLM Evaluation",
  "summary": "Benchmark datasets are central to evaluating Large Language Models (LLMs), yet they are typically conceived as monolithic tasks, obscuring substantial variation in the demands of individual samples. We introduce a dataset-centric meta-evaluation framework that audits benchmark datasets at the sample level along five latent dimensions: 1. Cognitive and Knowledge Demands, 2. Language and Content Quality, 3. Task Properties, 4. Context, and 5. Ethics, Safety, and Fairness. Applying this framework,",
  "authors": "Philipp D. Siedler, Jordan Sassoon",
  "category": "research",
  "topics": "bias-fairness,transparency",
  "orgs": null,
  "regions": null,
  "published_at": "2026-07-30T19:51:55.000Z",
  "fetched_at": "2026-08-03T05:10:47.622Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/15692",
  "original_url": "https://arxiv.org/abs/2607.28801v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}