{
  "id": 5660,
  "url": "https://arxiv.org/abs/2604.17457v4",
  "title": "Beyond the Bellman Fixed Point: Geometry and Fast Policy Identification in Value Iteration",
  "summary": "Q-value iteration (Q-VI) is usually analyzed through the \\(γ\\)-contraction of the Bellman operator. This argument proves convergence to \\(Q^*\\), but it gives only a coarse account of when the induced greedy policy becomes optimal. We study discounted Q-VI as a switching system and focus on the practically optimal solution set (POSS), the set of \\(Q\\)-functions whose tie-broken greedy policies are optimal. The main result shows that Q-VI reaches the optimal action class in finite time by entering",
  "authors": "Donghwan Lee",
  "category": "research",
  "topics": "regulation",
  "orgs": null,
  "regions": null,
  "published_at": "2026-04-19T14:18:18.000Z",
  "fetched_at": "2026-07-14T16:31:53.169Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/5660",
  "original_url": "https://arxiv.org/abs/2604.17457v4",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}