{
  "id": 6540,
  "url": "https://arxiv.org/abs/2604.00072v1",
  "title": "Empirical Validation of the Classification-Verification Dichotomy for AI Safety Gates",
  "summary": "Can classifier-based safety gates maintain reliable oversight as AI systems improve over hundreds of iterations? We provide comprehensive empirical evidence that they cannot. On a self-improving neural controller (d=240), eighteen classifier configurations -- spanning MLPs, SVMs, random forests, k-NN, Bayesian classifiers, and deep networks -- all fail the dual conditions for safe self-improvement. Three safe RL baselines (CPO, Lyapunov, safety shielding) also fail. Results extend to MuJoCo benc",
  "authors": "Arsenios Scrivens",
  "category": "research",
  "topics": "safety-alignment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-03-31T13:54:36.000Z",
  "fetched_at": "2026-07-14T16:32:33.102Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/6540",
  "original_url": "https://arxiv.org/abs/2604.00072v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}