{
  "id": 5065,
  "url": "https://arxiv.org/abs/2605.01643v3",
  "title": "AI Alignment via Incentives and Correction",
  "summary": "We study AI alignment through the lens of law-and-economics models of deterrence and enforcement. In these models, misconduct is not treated as an external failure, but as a strategic response to incentives: an actor weighs the gain from violation against the probability of detection and the severity of punishment. We argue that the same logic arises naturally in agentic AI pipelines. A solver may benefit from producing a persuasive but incorrect answer, hiding uncertainty, or exploiting spuriou",
  "authors": "Rohit Agarwal, Joshua Lin, Mark Braverman, Elad Hazan",
  "category": "research",
  "topics": "regulation,safety-alignment,agents-autonomy",
  "orgs": null,
  "regions": null,
  "published_at": "2026-05-02T23:28:02.000Z",
  "fetched_at": "2026-07-14T16:31:31.208Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/5065",
  "original_url": "https://arxiv.org/abs/2605.01643v3",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}