{
  "id": 18751,
  "url": "https://arxiv.org/abs/2608.11274",
  "title": "Agent Safety Should Be a Runtime Contract",
  "summary": "The dominant paradigm treats AI safety as a property to be instilled during model training via RLHF, DPO, or Constitutional AI. We argue this is structurally insufficient for autonomous agents that execute code, mutate files, send messages, and modify databases. Agent safety should be a runtime contract enforced by the harness, and the contract has two complementary faces. The preventive face blocks dangerous actions before they happen via sandboxes, permission gates, output filters, and traject",
  "authors": "Albus W. Ng, Yi Han, Jusheng Zhang, Wenhao Wang",
  "category": "research",
  "topics": "safety-alignment,agents-autonomy",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-10T20:00:00.000Z",
  "fetched_at": "2026-08-13T05:10:37.786Z",
  "source_slug": "hf-daily",
  "source_name": "HuggingFace Daily Papers",
  "source_homepage": "https://huggingface.co/papers",
  "ethics_ai_record_url": "https://ethics.ai/record/18751",
  "original_url": "https://arxiv.org/abs/2608.11274",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}