{
  "id": 502,
  "url": "https://arxiv.org/abs/2606.28739v1",
  "title": "Agent Safety Is Action Alignment",
  "summary": "Large language models increasingly act as agents: they call tools, move money, delete records, and send messages on a user's behalf. To keep them safe, practitioners imported the chatbot-era recipe (train the model to refuse unsafe inputs) into the agentic setting, and treat the resulting capability loss as a manageable ``alignment tax.'' We argue this is a \\emph{category error}. Refusal is a primitive for \\emph{content safety}, where the harm is in the model's output and is therefore a learnabl",
  "authors": "Shawn Li, Yue Zhao",
  "category": "research",
  "topics": "safety-alignment,agents-autonomy",
  "orgs": null,
  "regions": null,
  "published_at": "2026-06-27T05:26:43.000Z",
  "fetched_at": "2026-07-14T14:14:37.244Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/502",
  "original_url": "https://arxiv.org/abs/2606.28739v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}