{
  "id": 18255,
  "url": "https://arxiv.org/abs/2608.09826v1",
  "title": "Distill Skills into Weights, Not Prompts: Abstract Skills as Privileged Signals for On-Policy Self-Distillation",
  "summary": "Reinforcement learning with verifiable rewards yields no group-relative signal when rollout groups are uniformly correct or uniformly wrong, which account for 63.0-68.0% of groups in our experiments. We propose SKALD (Skill-Anchored Latent Distillation), an on-policy self-distillation framework that uses two context views of the same Qwen3-Base model: a question-only student and a teacher conditioned on an abstract, explicit-answer-filtered skill card. The student is trained on its own prefixes,",
  "authors": "Yubo Jiang, Fengying Xie, Zhiguo Jiang, Haopeng Zhang",
  "category": "research",
  "topics": "regulation,children-education",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-10T16:43:18.000Z",
  "fetched_at": "2026-08-11T05:10:37.351Z",
  "source_slug": "x-arxiv-cs-ai",
  "source_name": "arXiv cs.AI",
  "source_homepage": "https://arxiv.org/list/cs.AI/recent",
  "ethics_ai_record_url": "https://ethics.ai/record/18255",
  "original_url": "https://arxiv.org/abs/2608.09826v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}