{
  "id": 18314,
  "url": "https://arxiv.org/abs/2608.08471v1",
  "title": "Yesterday's Shield, Today's Spear: A Self-Evolving Safety Guardrail in Production",
  "summary": "Deployed LLM safety guardrails are predominantly static: trained once and frozen at release, while new jailbreak techniques and previously un-addressed harmful categories emerge within days, leaving the defense perpetually a step behind. We present SESG (Self-Evolving Safety Guardrails), a multi-agent system running in production. SESG monitors the live traffic behind a deployed guardrail and surfaces two classes of failure: jailbreaks novel in form and harmful categories novel in content. Once",
  "authors": "Cong Ming, Jingyi Chen, Bin Liu, Qi Chu, Tao Gong, Nenghai Yu, Yingfei Xiang",
  "category": "research",
  "topics": "safety-alignment,military-security,agents-autonomy",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-09T04:31:05.000Z",
  "fetched_at": "2026-08-11T05:10:37.351Z",
  "source_slug": "x-arxiv-red-teaming-query",
  "source_name": "arXiv red teaming query",
  "source_homepage": "https://arxiv.org/a/redteam",
  "ethics_ai_record_url": "https://ethics.ai/record/18314",
  "original_url": "https://arxiv.org/abs/2608.08471v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}