{
  "id": 18327,
  "url": "https://arxiv.org/abs/2608.08764v1",
  "title": "Learning from Consensus and Disagreement: Unsupervised On-Policy Self-Distillation with Minority-Trajectory Contrast",
  "summary": "On-policy self-distillation improves language-model reasoning by querying a teacher on states actually visited by the student. Recent methods create a powerful information asymmetry by exposing the teacher to privileged context, yet they fundamentally rely on external supervision---such as gold solutions or verifiers---to construct this advantage. We introduce CoDA (Consensus and Disagreement Alignment), a fully unsupervised framework that creates reliable privileged information entirely from th",
  "authors": "Jiaxin Guo, Yanwei Yue, Xuanbo Fan, Chunyu Yang, Yan Zhang",
  "category": "research",
  "topics": "regulation,safety-alignment,children-education",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-09T15:23:25.000Z",
  "fetched_at": "2026-08-11T05:10:37.351Z",
  "source_slug": "arxiv-cslg",
  "source_name": "arXiv cs.LG",
  "source_homepage": "https://arxiv.org/list/cs.LG/recent",
  "ethics_ai_record_url": "https://ethics.ai/record/18327",
  "original_url": "https://arxiv.org/abs/2608.08764v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}