{
  "id": 19070,
  "url": "https://arxiv.org/abs/2608.11560v1",
  "title": "When Offline Evaluation Misleads: A Diagnostic Protocol for Reward and Policy Selection in Delayed-Feedback Contextual Bandits",
  "summary": "Personalizing marketing messages with contextual multi-armed bandits (CMABs) drives real business value, yet the objective that ultimately matters - a downstream conversion - is observed only weeks later, too late to drive online learning. Teams therefore train the bandit on a fast proxy reward, and separately must judge whether a contextual bandit is worth its complexity over sending one best message. Settling both decisions with the usual offline checks - a batch off-policy estimate, a margina",
  "authors": "Sang Su Lee, Vineeth Loganathan, Shishir Dash, Vijay Raghavan",
  "category": "research",
  "topics": "regulation,healthcare",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-12T01:53:21.000Z",
  "fetched_at": "2026-08-13T05:10:37.786Z",
  "source_slug": "arxiv-cslg",
  "source_name": "arXiv cs.LG",
  "source_homepage": "https://arxiv.org/list/cs.LG/recent",
  "ethics_ai_record_url": "https://ethics.ai/record/19070",
  "original_url": "https://arxiv.org/abs/2608.11560v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}