{
  "id": 14154,
  "url": "https://arxiv.org/abs/2607.25398v1",
  "title": "HANDBOOK.md: A Benchmark for Long-Context Agentic Instruction Following",
  "summary": "Language-model agents are increasingly deployed under standing instructions: a system prompt, a policy file, or a skills document is placed in context, and the agent is trusted to let it govern every action that follows. Existing benchmarks rarely test this deployment pattern directly; they measure whether an agent can complete a task, not whether a long, binding policy document actually constrains its behavior over an extended tool-use horizon. We present HANDBOOK.md, a benchmark of 65 agentic",
  "authors": "Liudas Panavas, Sebastian Minus, Bradley Monton, Derek Ray, Suhaas Garre, Sushant Mehta et al.",
  "category": "research",
  "topics": "regulation,agents-autonomy",
  "orgs": null,
  "regions": null,
  "published_at": "2026-07-28T07:58:07.000Z",
  "fetched_at": "2026-07-29T05:10:12.205Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/14154",
  "original_url": "https://arxiv.org/abs/2607.25398v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}