{
  "id": 16126,
  "url": "https://arxiv.org/abs/2608.01042v1",
  "title": "What Could the Agent See at 19:05? Generating Temporal Enterprise Scenarios from Real Research and Replaying Them to Evaluate Agents",
  "summary": "Enterprise AI agents act across many apps whose data changes continuously, so an answer is correct only relative to what data existed and who could see it at the moment it was asked. Offline evaluation today grades against a single static snapshot, effectively the end of the episode. So, it can only evaluate one situation, the final one, even though every earlier moment of the episode is a different situation that invites its own realistic questions with its own correct answers. Recreating each",
  "authors": "Tezan Sahu, Himani Arora",
  "category": "research",
  "topics": "agents-autonomy",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-02T06:56:56.000Z",
  "fetched_at": "2026-08-04T05:10:21.797Z",
  "source_slug": "x-arxiv-cs-hc",
  "source_name": "arXiv cs.HC",
  "source_homepage": "https://arxiv.org/list/cs.HC/recent",
  "ethics_ai_record_url": "https://ethics.ai/record/16126",
  "original_url": "https://arxiv.org/abs/2608.01042v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}