{
  "id": 16718,
  "url": "https://www.frontiersin.org/articles/10.3389/frai.2026.1892739",
  "title": "AdaK: adaptive KV cache budget estimation framework for analyzing long-context large language model inference",
  "summary": "IntroductionThe deployment of LLMs on resource-constrained hardware is hindered by the memory-intensive KV Cache mechanism.MethodsWe propose AdaK, an adaptive KV cache budget estimation framework with three strategies: entropy-based thresholding, task-aware lookup table, and a lightweight policy network.ResultsAdaK reveals estimated KV cache reductions of up to 17.9% relative to fixed-k = 2048 baselines across 16 settings on Qwen3-4B, Qwen3-8B, and Mistral-7B.DiscussionAdaK's decoupled design en",
  "authors": "Tianjun Shao",
  "category": "research",
  "topics": "regulation",
  "orgs": "mistral",
  "regions": null,
  "published_at": "2026-08-05T00:00:00.000Z",
  "fetched_at": "2026-08-06T05:10:11.148Z",
  "source_slug": "x-frontiers-in-artificial-intelligence",
  "source_name": "Frontiers in Artificial Intelligence",
  "source_homepage": "https://www.frontiersin.org/journals/artificial-intelligence",
  "ethics_ai_record_url": "https://ethics.ai/record/16718",
  "original_url": "https://www.frontiersin.org/articles/10.3389/frai.2026.1892739",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}