{
  "id": 10610,
  "url": "https://arxiv.org/abs/2607.13205v1",
  "title": "Adaptive Filtering of the KV Cache: Diagnosing and Correcting Structural-Role Bias in LLM Inference",
  "summary": "Attention-based KV cache eviction (H2O and its descendants) compresses the memory-constrained state of a long-context model by ranking tokens on accumulated attention mass, treated here as signal energy, and keeping the heaviest. On schema-dense input streams such as nested JSON, this score acts as a non-stationary filter that disproportionately retains noise: a non-content sink role (delimiters or whitespace) carries an order of magnitude more energy than any content role, and structural KEY to",
  "authors": "Soumil Mandal",
  "category": "research",
  "topics": "bias-fairness,healthcare,environment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-07-14T18:55:20.000Z",
  "fetched_at": "2026-07-16T05:10:56.605Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/10610",
  "original_url": "https://arxiv.org/abs/2607.13205v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}