{
  "id": 497,
  "url": "https://arxiv.org/abs/2606.28863v1",
  "title": "Defeat Devices in AI Systems",
  "summary": "AI systems increasingly exhibit behavior that differs systematically between evaluation and deployment contexts. Alignment faking, sandbagging, benchmark gaming, deceptive scheming, specification gaming, and trojans have each been documented separately, with each line of work characterizing one facet of what we argue is a single structural mechanism. We propose that this common mechanism is a defeat device, an engineering and regulatory concept long established in vehicle-emissions law and broug",
  "authors": "Emilio Ferrara",
  "category": "research",
  "topics": "regulation,safety-alignment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-06-27T11:12:17.000Z",
  "fetched_at": "2026-07-14T14:14:32.650Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/497",
  "original_url": "https://arxiv.org/abs/2606.28863v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}