{
  "id": 10560,
  "url": "https://arxiv.org/abs/2607.13124",
  "title": "ShortOPD: Recovering Pruned LLMs with Short-to-Long On-Policy Distillation",
  "summary": "Structured pruning is a hardware-friendly way to compress LLMs, but it is mostly validated on multiple-choice recognition tasks, while the same compressed checkpoints can collapse on the free-form generation that deployment actually requires. Two observations trace this gap. First, greedy pass@1 nearly vanishes after compression, yet pass@k recovers substantially under repeated sampling: useful generations are demoted, not erased. Second, the recoverable regime fails mainly through suffix repeti",
  "authors": "Qingyu Zhang, Qianhao Yuan, Hongyu Lin, Yaojie Lu, Xianpei Han, Le Sun",
  "category": "research",
  "topics": "regulation",
  "orgs": null,
  "regions": null,
  "published_at": "2026-07-13T20:00:00.000Z",
  "fetched_at": "2026-07-16T05:10:56.605Z",
  "source_slug": "hf-daily",
  "source_name": "HuggingFace Daily Papers",
  "source_homepage": "https://huggingface.co/papers",
  "ethics_ai_record_url": "https://ethics.ai/record/10560",
  "original_url": "https://arxiv.org/abs/2607.13124",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}