{
  "id": 18687,
  "url": "https://arxiv.org/abs/2608.10669v1",
  "title": "REDAgentBench: Executable Red Teaming and Faithful Measurement of LLM Agent Systems",
  "summary": "Large language model (LLM) agents combine language-based reasoning with external tools to perform complex tasks. Adversarial inputs can exploit interactions between the agent and its environment, causing the agent to violate safety policies during execution. Yet existing evaluations often reduce agent safety to a single attack success rate (ASR), collapsing exposure, execution, observation, and adjudication and potentially conflating actual violations with evidence visibility. We introduce REDAg",
  "authors": "Zixing Chen, Xingyuan Liu, Jie Zhu, Huaixia Dou, Shuo Jiang, Junhui Li, Lifan Guo, Feng Chen, Chi Zhang",
  "category": "research",
  "topics": "safety-alignment,agents-autonomy,environment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-11T08:48:54.000Z",
  "fetched_at": "2026-08-12T05:10:43.828Z",
  "source_slug": "x-arxiv-red-teaming-query",
  "source_name": "arXiv red teaming query",
  "source_homepage": "https://arxiv.org/a/redteam",
  "ethics_ai_record_url": "https://ethics.ai/record/18687",
  "original_url": "https://arxiv.org/abs/2608.10669v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}