{
  "id": 16094,
  "url": "https://arxiv.org/abs/2608.02464v1",
  "title": "Real-Time Detection and Repair of LLM Agent Failures",
  "summary": "LLM agents fail mid-episode -- they loop, cascade tool errors, drift off goal, fabricate results, or silently absorb corrupted content -- and the standard remedy, judging every step with a second LLM, costs more than the agent itself. We ask how much detection is achievable from observable step telemetry alone, using monitors costing microseconds per step and trained only on healthy runs. On 2,823 committed agent episodes across three frameworks, three local models (qwen2.5 7b/3b, llama3.1 8b) a",
  "authors": "Sunny Dubey",
  "category": "research",
  "topics": "healthcare,agents-autonomy",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-03T16:34:46.000Z",
  "fetched_at": "2026-08-04T05:10:21.797Z",
  "source_slug": "x-arxiv-cs-ai",
  "source_name": "arXiv cs.AI",
  "source_homepage": "https://arxiv.org/list/cs.AI/recent",
  "ethics_ai_record_url": "https://ethics.ai/record/16094",
  "original_url": "https://arxiv.org/abs/2608.02464v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}