{
  "id": 10352,
  "url": "https://arxiv.org/abs/2607.12188v1",
  "title": "Cost-Governed RAG: Unified Per-Tenant Cost Attribution Across Retrieval and Generation in Multi-Tenant LLM Systems",
  "summary": "Enterprise Retrieval-Augmented Generation (RAG) deployments face a critical governance gap: while LLM generation cost is metered per token, the retrieval layer - vector memory, similarity compute, and embedding API calls - remains an unattributed shared cost, enabling invisible cross-subsidization among tenants. We present Cost-Governed RAG, an architecture that integrates a codebook-oblivious vector index (TurboVec) with a multi-tenant LLM governance gateway, creating a unified observability st",
  "authors": "Navnit Shukla",
  "category": "research",
  "topics": "regulation",
  "orgs": null,
  "regions": null,
  "published_at": "2026-07-13T22:16:58.000Z",
  "fetched_at": "2026-07-15T05:10:55.633Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/10352",
  "original_url": "https://arxiv.org/abs/2607.12188v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}