{
  "id": 7441,
  "url": "https://arxiv.org/abs/2603.09378v2",
  "title": "SPAARS: Safer RL Policy Alignment through Abstract Exploration and Refined Exploitation of Action Space",
  "summary": "Offline-to-online reinforcement learning (RL) offers a promising paradigm for robotics by pre-training policies on safe, offline demonstrations and fine-tuning them via online interaction. However, a fundamental challenge remains: how to safely explore online without deviating from the behavioral support of the offline data? While recent methods leverage conditional variational autoencoders (CVAEs) to bound exploration within a latent space, they inherently suffer from an exploitation gap -- a p",
  "authors": "Swaminathan S K, Aritra Hazra",
  "category": "research",
  "topics": "regulation,safety-alignment,agents-autonomy",
  "orgs": null,
  "regions": null,
  "published_at": "2026-03-10T08:52:15.000Z",
  "fetched_at": "2026-07-14T16:33:12.393Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/7441",
  "original_url": "https://arxiv.org/abs/2603.09378v2",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}