{
  "id": 19476,
  "url": "https://arxiv.org/abs/2608.12489v1",
  "title": "When Can You Trust Offline Evaluation of Equal-Cost Top-k Allocation? A Controlled, Reproducible Benchmark and Practitioner's Guide",
  "summary": "Organizations decide whom to treat under a budget and want to know what a targeting rule would have earned before deploying it. Off-policy evaluation promises this from logged data, but the deployable rule is a deterministic top-k policy: it removes all averaging over actions, so weak overlap hits the estimate directly. We benchmark six estimators across five datasets and two known-effect sweeps, and validate the mechanisms against a non-simulated paired reference. First, weak overlap is governe",
  "authors": "Binshuang Li",
  "category": "research",
  "topics": "regulation",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-12T18:10:10.000Z",
  "fetched_at": "2026-08-14T05:10:49.168Z",
  "source_slug": "arxiv-cslg",
  "source_name": "arXiv cs.LG",
  "source_homepage": "https://arxiv.org/list/cs.LG/recent",
  "ethics_ai_record_url": "https://ethics.ai/record/19476",
  "original_url": "https://arxiv.org/abs/2608.12489v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}