{
  "id": 3396,
  "url": "https://arxiv.org/abs/2606.20634v1",
  "title": "DEMM-Bench: A Cross-Regime Benchmark for Agent-Runtime Governance-Evidence Sufficiency",
  "summary": "Agent-runtime systems emit traces, ledgers, provenance graphs, policy logs, delegation tokens, cache events, and tool-firewall records, but those containers do not necessarily answer governance questions about a specific decision. DEMM-Bench is a cross-regime benchmark for agent-runtime governance-evidence sufficiency, grounded in the Decision Evidence Maturity Model (DEMM): it measures whether records across eight evidence regimes are sufficient to reconstruct decision-level properties rather t",
  "authors": "Oleg Solozobov",
  "category": "research",
  "topics": "regulation,agents-autonomy",
  "orgs": null,
  "regions": null,
  "published_at": "2026-05-30T09:36:29.000Z",
  "fetched_at": "2026-07-14T16:30:14.369Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/3396",
  "original_url": "https://arxiv.org/abs/2606.20634v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}