{
  "id": 15793,
  "url": "https://arxiv.org/abs/2607.28685v1",
  "title": "Safety, or Just Capability? A Validity Audit of Agent-Safety Benchmarks",
  "summary": "Agent-safety benchmarks measure different behaviors, and their scores get quoted interchangeably as an agent's safety. We treat four of them (R-Judge, InjecAgent, AgentHarm, AgentDojo) as measurements to be validated, running each under its official implementation and author-provided scorer on up to 22 models, with MMLU and GPQA measured by us under one protocol as a capability composite. The metric is the first problem. On any binary trace-judgment benchmark scored by $F_1$, an ``always positiv",
  "authors": "Youting Wang, Xiao Han, Dingyan Shang, Yuan Tang, Bowen Liu",
  "category": "research",
  "topics": "agents-autonomy,transparency",
  "orgs": null,
  "regions": null,
  "published_at": "2026-07-30T03:45:10.000Z",
  "fetched_at": "2026-08-03T05:10:47.622Z",
  "source_slug": "x-arxiv-red-teaming-query",
  "source_name": "arXiv red teaming query",
  "source_homepage": "https://arxiv.org/a/redteam",
  "ethics_ai_record_url": "https://ethics.ai/record/15793",
  "original_url": "https://arxiv.org/abs/2607.28685v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}