{
  "id": 4700,
  "url": "https://arxiv.org/abs/2605.08442v4",
  "title": "Defense effectiveness across architectural layers: a mechanistic evaluation of persistent memory attacks on stateful LLM agents",
  "summary": "Persistent memory in LLM agents creates an attack surface that production safety classifiers do not observe: the payload enters via RAG retrieval and persists across sessions via tool-mediated memory. We evaluate six defenses across four architectural layers against delayed-trigger attacks on nine open-source models (5,040 runs, N=40 per condition). Five of six defenses fail: input-level filters never see the payload (it enters via RAG, not user input); retrieval-level classifiers observe it but",
  "authors": "Jun Wen Leong",
  "category": "research",
  "topics": "military-security,agents-autonomy",
  "orgs": null,
  "regions": null,
  "published_at": "2026-05-08T20:04:57.000Z",
  "fetched_at": "2026-07-14T16:31:12.744Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/4700",
  "original_url": "https://arxiv.org/abs/2605.08442v4",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}