{
  "id": 16220,
  "url": "https://arxiv.org/abs/2607.26451",
  "title": "ExplainBench: Evaluating Code Explanations from Agents",
  "summary": "Large Language Model (LLM) agents have seen rapid adoption in software engineering. As agents take a greater role in the actual generation of code, they are making larger changes, spanning tens to hundreds of lines. This makes manual review of agent results increasingly infeasible, leading developers to turn to explanations to understand enacted changes. Despite this, there are no benchmarks that evaluate the trustworthiness of agent-generated explanations. To bridge this gap, we propose Explain",
  "authors": "Zhiyuan Pan, Sungmin Kang, Imam Nur Bani Yusuf, Abhik Roychoudhury",
  "category": "research",
  "topics": "agents-autonomy",
  "orgs": null,
  "regions": null,
  "published_at": "2026-07-28T20:00:00.000Z",
  "fetched_at": "2026-08-05T05:10:44.550Z",
  "source_slug": "hf-daily",
  "source_name": "HuggingFace Daily Papers",
  "source_homepage": "https://huggingface.co/papers",
  "ethics_ai_record_url": "https://ethics.ai/record/16220",
  "original_url": "https://arxiv.org/abs/2607.26451",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}