{
  "id": 648,
  "url": "https://arxiv.org/abs/2606.24622v1",
  "title": "Themis: An explainable AI-enabled framework for Reinforcement Learning with Human Feedback",
  "summary": "Training safe Reinforcement Learning (RL) systems is inherently challenging, with no guarantee of avoiding unwanted behaviors. The most effective defenses against this are (i) transparency through explainability and (ii) alignment via human feedback. While both show promising results, no publicly available framework currently combines them. To address this, we introduce Themis, an XAI-enabled testing and evaluation framework for Reinforcement Learning from Human Feedback. Themis supports over 20",
  "authors": "Andreas Chouliaras, Luke Connolly, Dimitris Chatzpoulos",
  "category": "research",
  "topics": "safety-alignment,transparency",
  "orgs": null,
  "regions": null,
  "published_at": "2026-06-23T14:20:42.000Z",
  "fetched_at": "2026-07-14T14:14:41.551Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/648",
  "original_url": "https://arxiv.org/abs/2606.24622v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}