{
  "id": 1457,
  "url": "https://arxiv.org/abs/2606.05682v2",
  "title": "Beyond Output Matching: Preserving Internal Geometry in NVFP4 LLM Distillation",
  "summary": "Demand for low-precision inference, including NVFP4-based approaches, has grown as large language models are increasingly deployed in latency and cost constrained production environments. Quantization-aware distillation (QAD) helps recover accuracy lost under low bit quantization by training a quantized student to match the output distribution of a frozen higher precision teacher via a KL-divergence loss. In this work, we first provide a representation level diagnosis of QAD: output matching alo",
  "authors": "Fangbo Tu, Junhua Zhao, Chi Liu, Xin Chen, Haifeng Wu, Jian Wan et al.",
  "category": "research",
  "topics": "healthcare,children-education,environment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-06-04T04:03:42.000Z",
  "fetched_at": "2026-07-14T14:15:17.102Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/1457",
  "original_url": "https://arxiv.org/abs/2606.05682v2",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}