{
  "id": 7263,
  "url": "https://arxiv.org/abs/2603.15676v2",
  "title": "Automated Self-Testing as a Quality Gate: Evidence-Driven Release Management for LLM Applications",
  "summary": "LLM applications are AI systems whose nondeterministic outputs and evolving model behavior make traditional testing insufficient for release governance. We present an automated self-testing framework that introduces quality gates with evidence-based release decisions (PROMOTE/HOLD/ROLLBACK) across five empirically grounded dimensions: task success rate, research context preservation, P95 latency, safety pass rate, and evidence coverage. We evaluate the framework through a longitudinal case study",
  "authors": "Alexandre Cristovão Maiorano",
  "category": "research",
  "topics": "regulation",
  "orgs": null,
  "regions": null,
  "published_at": "2026-03-13T20:44:15.000Z",
  "fetched_at": "2026-07-14T16:33:08.010Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/7263",
  "original_url": "https://arxiv.org/abs/2603.15676v2",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}