{
  "id": 19442,
  "url": "https://arxiv.org/abs/2608.13417v1",
  "title": "Beyond Final Scores: A Systematic Evaluation of Agents for Long-Horizon AI Research and Development",
  "summary": "Autonomous agents are increasingly capable of improving models, systems, and other technical artifacts through long-horizon experimentation. To understand the current state of this capability, however, evaluation must go beyond final scores, which neither reveal where progress is gained or lost nor indicate whether accumulated experience improves later decisions. We therefore present a systematic evaluation of seven frontier models on 36 long-horizon tasks based on a new framework that uses rule",
  "authors": "Yiwei Li, Wanli Yang, Hexiang Tan, Xiangzhou Huang, Zhengyu Chen, Ziran Li, Borun Chen, Shanglin Lei, Huaisheng Zhu, Hao Tian, Fei Sun, Xunliang Cai, Jingang Wang",
  "category": "research",
  "topics": "agents-autonomy",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-13T16:11:22.000Z",
  "fetched_at": "2026-08-14T05:10:49.168Z",
  "source_slug": "x-arxiv-cs-ai",
  "source_name": "arXiv cs.AI",
  "source_homepage": "https://arxiv.org/list/cs.AI/recent",
  "ethics_ai_record_url": "https://ethics.ai/record/19442",
  "original_url": "https://arxiv.org/abs/2608.13417v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}