{
  "id": 3562,
  "url": "https://arxiv.org/abs/2605.29028v1",
  "title": "Return-to-Go Is More Than a Number: Q-Guided Alignment for Return-Conditioned Supervised Learning",
  "summary": "Conditioned Sequence Models (CSMs) learn policies by treating return-to-go (RTG) as a control signal. However, existing CSMs often treat the RTGs as simple numerical inputs rather than aligning them with the performance of their policies. In this paper, we propose Q-ALIGN DT, a framework that enforces this alignment by ensuring the $Q$-value of the output policy is consistent with the input RTG. By leveraging a $Q$ function to provide dense guidance to CSMs and further fine-tuning it using an RT",
  "authors": "Yuxiao Yang, Weitong Zhang",
  "category": "research",
  "topics": "regulation,safety-alignment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-05-27T19:24:35.000Z",
  "fetched_at": "2026-07-14T16:30:18.858Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/3562",
  "original_url": "https://arxiv.org/abs/2605.29028v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}