{
  "id": 16655,
  "url": "https://arxiv.org/abs/2608.04830v1",
  "title": "ContextWeave: A Real-World Workflow Benchmark",
  "summary": "Memory is essential as language agents move from isolated tasks to long-horizon, stateful workflows, yet existing evaluations often reduce it to retrieval or question answering. We introduce ContextWeave, a longitudinal benchmark that evaluates whether recalled experience improves downstream agent performance in realistic office-work streams. ContextWeave reconstructs privacy-preserved, multi-month workflows of 14 participants into 1,005 executable tasks, including 568 core evaluation tasks, wit",
  "authors": "Bo Wang, Yuqian Yao, Enxi Wang, Luozhijie Jin, Yang Liu, Yiran Suo et al.",
  "category": "research",
  "topics": "privacy-surveillance,agents-autonomy",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-05T13:31:18.000Z",
  "fetched_at": "2026-08-06T05:10:11.148Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/16655",
  "original_url": "https://arxiv.org/abs/2608.04830v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}