{
  "id": 13699,
  "url": "https://arxiv.org/abs/2607.22368v1",
  "title": "Do Agent Benchmarks Measure Capability? Protocol Validity in the Age of Agentic AI",
  "summary": "Agent benchmarks increasingly evaluate repository editing, web research, terminal use, and long-horizon interaction. Their scores support capability claims only when the evaluation protocol keeps the intended capability necessary for success. Recent reward-hacking benchmarks and system reports show that agents can instead recover public solutions, read evaluation artifacts, infer generator structure, manipulate feedback, or benefit from invalid scoring paths; existing responses do not provide a",
  "authors": "Jiaqi Shao, Hanck Chen, Wei Zhang, Maxm Pan, Bing Luo",
  "category": "research",
  "topics": "agents-autonomy",
  "orgs": null,
  "regions": null,
  "published_at": "2026-07-24T14:55:19.000Z",
  "fetched_at": "2026-07-27T05:10:06.638Z",
  "source_slug": "x-arxiv-cs-ai",
  "source_name": "arXiv cs.AI",
  "source_homepage": "https://arxiv.org/list/cs.AI/recent",
  "ethics_ai_record_url": "https://ethics.ai/record/13699",
  "original_url": "https://arxiv.org/abs/2607.22368v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}