{
  "id": 5346,
  "url": "https://arxiv.org/abs/2604.23646v1",
  "title": "Structural Enforcement of Goal Integrity in AI Agents via Separation-of-Powers Architecture",
  "summary": "Recent evidence suggests that frontier AI systems can exhibit agentic misalignment, generating and executing harmful actions derived from internally constructed goals, even without explicit user requests. Existing mitigation methods, such as Reinforcement Learning from Human Feedback (RLHF) and constitutional prompting, operate primarily at the model level and provide only probabilistic safety guarantees. We propose the Policy-Execution-Authorization (PEA) architecture, a \"separation-of-powers\" ",
  "authors": "Rong Xiang",
  "category": "research",
  "topics": "regulation,safety-alignment,agents-autonomy",
  "orgs": null,
  "regions": null,
  "published_at": "2026-04-26T10:31:13.000Z",
  "fetched_at": "2026-07-14T16:31:40.222Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/5346",
  "original_url": "https://arxiv.org/abs/2604.23646v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}