{
  "id": 17935,
  "url": "https://arxiv.org/abs/2608.06557v1",
  "title": "Cascade: Exploiting SLO-Aware latency budget for fair and high goodput LLM inference serving",
  "summary": "The reasoning and agentic capabilities of large language models have expanded the range of applications they support, from short interactive exchanges to long, compute-heavy requests. LLM serving platforms today define response-latency service-level objectives, even though requests within the same service can differ by orders of magnitude in input length, generation length, execution cost, and the availability of reusable KV-cache state. As a result, requests governed by the same service level o",
  "authors": "Muhammad Adnan, Rohan Mahapatra, Prashant J. Nair, Daniel Berger, Pantea Zardoshti, Rodrigo Fonseca, Esha Choukse",
  "category": "research",
  "topics": "agents-autonomy",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-06T20:14:21.000Z",
  "fetched_at": "2026-08-10T05:10:00.488Z",
  "source_slug": "x-arxiv-fairness-query",
  "source_name": "arXiv fairness query",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/17935",
  "original_url": "https://arxiv.org/abs/2608.06557v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}