{
  "id": 14889,
  "url": "https://arxiv.org/abs/2607.28022",
  "title": "Flux-OPD: On-Policy Distillation with Evolving Contexts",
  "summary": "Large language model training in open-ended domains lacks verifiable rewards, making task preferences difficult to formalize as effective supervision. Contexts can convey such preferences, yet provide little additional supervision once distilled into the student, motivating contexts that evolve with student performance. However, directly using evolving contexts as in-training supervision results in an unstable distillation target and conflicting distributions, requiring mechanisms to stabilize t",
  "authors": "Yuran Wang, Zekun Wang, Bohan Zeng, Ruixu Zhang, Wenxuan Liu, Liu Yang",
  "category": "research",
  "topics": "regulation,children-education",
  "orgs": null,
  "regions": null,
  "published_at": "2026-07-29T20:00:00.000Z",
  "fetched_at": "2026-07-31T05:10:57.675Z",
  "source_slug": "hf-daily",
  "source_name": "HuggingFace Daily Papers",
  "source_homepage": "https://huggingface.co/papers",
  "ethics_ai_record_url": "https://ethics.ai/record/14889",
  "original_url": "https://arxiv.org/abs/2607.28022",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}