{
  "id": 1461,
  "url": "https://arxiv.org/abs/2606.05670v1",
  "title": "Do More Agents Help? Controlled and Protocol-Aligned Evaluation of LLM Agent Workflows",
  "summary": "Does adding more agents help an LLM workflow once compared systems share the same benchmark loader, tool access, answer contract, usage accounting, and trajectory logging? We introduce BenchAgent, an evaluation framework that places single-agent, fixed multi-agent (MAS), and evolving MAS workflows under one normalized execution and logging protocol. BenchAgent evaluates these substrate-internal workflows across ten reasoning, coding, and tool-use benchmarks with GPT-4.1, and separately reports a",
  "authors": "Yuhang Fu, Ruishan Fang, Jiaqi Shao, Huiyu Zheng, Zhengtao Zhu, Bing Luo et al.",
  "category": "research",
  "topics": "agents-autonomy",
  "orgs": "openai",
  "regions": null,
  "published_at": "2026-06-04T03:50:47.000Z",
  "fetched_at": "2026-07-14T14:15:17.103Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/1461",
  "original_url": "https://arxiv.org/abs/2606.05670v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}