{
  "id": 5153,
  "url": "https://arxiv.org/abs/2604.28049v1",
  "title": "Agent-Agnostic Evaluation of SQL Accuracy in Production Text-to-SQL Systems",
  "summary": "Text-to-SQL (T2SQL) evaluation in production environments poses fundamental challenges that existing benchmarks do not address. Current evaluation methodologies whether rule-based SQL matching or schema-dependent semantic parsers assume access to ground-truth queries and structured database schema, constraints that are rarely satisfied in real-world deployments. This disconnect leaves production T2SQL agents largely unevaluated beyond developer-time testing, creating silent quality degradation w",
  "authors": "Taslim Jamal Arif, Kuldeep Singh",
  "category": "research",
  "topics": "agents-autonomy,environment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-04-30T15:59:28.000Z",
  "fetched_at": "2026-07-14T16:31:31.213Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/5153",
  "original_url": "https://arxiv.org/abs/2604.28049v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}