{
  "id": 6513,
  "url": "https://arxiv.org/abs/2604.00324v1",
  "title": "The Persistent Vulnerability of Aligned AI Systems",
  "summary": "Autonomous AI agents are being deployed with filesystem access, email control, and multi-step planning. This thesis contributes to four open problems in AI safety: understanding dangerous internal computations, removing dangerous behaviors once embedded, testing for vulnerabilities before deployment, and predicting when models will act against deployers. ACDC automates circuit discovery in transformers, recovering all five component types from prior manual work on GPT-2 Small by selecting 68 edg",
  "authors": "Aengus Lynch",
  "category": "research",
  "topics": "safety-alignment,agents-autonomy",
  "orgs": "openai",
  "regions": null,
  "published_at": "2026-03-31T23:49:07.000Z",
  "fetched_at": "2026-07-14T16:32:33.100Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/6513",
  "original_url": "https://arxiv.org/abs/2604.00324v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}