{
  "id": 12310,
  "url": "https://arxiv.org/abs/2607.07050",
  "title": "Diagnosing and Calibrating Tool-Call Boundary Drift in Multi-Teacher On-Policy Distillation",
  "summary": "Agentic language models must learn when to call tools, when to consume tool responses, and when to answer directly. This makes multi-teacher on-policy distillation a natural training strategy: one teacher can specialize in tool calls, another in direct responses, and the student can learn from both on its own generated distribution. We show that this strategy can induce a behavior shift that is invisible from aggregate losses alone. In a two-teacher tool-use setting, vanilla generalized knowledg",
  "authors": "Jiabin Shen, Guang Chen, Chengjun Mao",
  "category": "research",
  "topics": "regulation,healthcare,children-education,agents-autonomy",
  "orgs": null,
  "regions": null,
  "published_at": "2026-07-14T20:00:00.000Z",
  "fetched_at": "2026-07-22T05:10:49.469Z",
  "source_slug": "hf-daily",
  "source_name": "HuggingFace Daily Papers",
  "source_homepage": "https://huggingface.co/papers",
  "ethics_ai_record_url": "https://ethics.ai/record/12310",
  "original_url": "https://arxiv.org/abs/2607.07050",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}