{
  "id": 17973,
  "url": "https://arxiv.org/abs/2608.09802",
  "title": "SWE-Bench ProMax: Benchmarking Agents on Large-Scale Multilingual Code Refactoring",
  "summary": "As AI coding agents take on increasingly complex, long-horizon software engineering tasks, existing benchmarks are rapidly saturating and their evaluation quality has come under serious scrutiny: a recent audit found that nearly 60% of unsolved SWE-bench Verified instances contain flawed tests -- either overly narrow tests that reject correct solutions or overly broad tests that check unstated requirements -- and that frontier models can verbatim reproduce gold patches from training data. Code r",
  "authors": "Yuling Shi, Jinghan Xu, Kelin Fu, Wenhao Zeng, Shilin He, Lei Zhang",
  "category": "research",
  "topics": "agents-autonomy,transparency",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-09T20:00:00.000Z",
  "fetched_at": "2026-08-11T05:10:37.351Z",
  "source_slug": "hf-daily",
  "source_name": "HuggingFace Daily Papers",
  "source_homepage": "https://huggingface.co/papers",
  "ethics_ai_record_url": "https://ethics.ai/record/17973",
  "original_url": "https://arxiv.org/abs/2608.09802",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}