{
  "id": 3018,
  "url": "https://arxiv.org/abs/2607.11720v1",
  "title": "Active Offline-to-Online Reinforcement Learning",
  "summary": "Background: Offline reinforcement learning (RL) enables effective policies to be trained from large, previously collected datasets and subsequently improved through limited online interaction. This offline-to-online RL (O2O-RL) paradigm is particularly promising in nonstationary domains where interaction is costly or potentially hazardous. Standard O2O-RL pipelines train multiple candidate policies offline, evaluate them using off-policy or online evaluation, and then deploy and fine-tune the po",
  "authors": "Alper Kamil Bozkurt, Shangtong Zhang, Yuichi Motai",
  "category": "research",
  "topics": "regulation",
  "orgs": null,
  "regions": null,
  "published_at": "2026-07-13T15:46:11.000Z",
  "fetched_at": "2026-07-14T16:11:46.979Z",
  "source_slug": "x-arxiv-cs-ai",
  "source_name": "arXiv cs.AI",
  "source_homepage": "https://arxiv.org/list/cs.AI/recent",
  "ethics_ai_record_url": "https://ethics.ai/record/3018",
  "original_url": "https://arxiv.org/abs/2607.11720v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}