{
  "id": 18758,
  "url": "https://arxiv.org/abs/2608.10636",
  "title": "DistilVDR: A Compact End-to-End Visual Document Retriever via Dual-Student Distillation",
  "summary": "Visual document retrieval (VDR) is dominated by multi-billion-parameter models that are slow to index at full corpus scale and expensive to serve. Prior compression routes either train a smaller multi-vector encoder from scratch or distil only the query side; neither yields a compact single-vector retriever end-to-end. We present DistilVDR, a 524M end-to-end VDR system distilled bilaterally from a single 8B vision-language teacher under a pointwise cosine alignment loss. All supervision comes fr",
  "authors": "Zhuchenyang Liu, Ziyi Wang, Yao Zhang, Yu Xiao",
  "category": "research",
  "topics": "safety-alignment,children-education",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-10T20:00:00.000Z",
  "fetched_at": "2026-08-13T05:10:37.786Z",
  "source_slug": "hf-daily",
  "source_name": "HuggingFace Daily Papers",
  "source_homepage": "https://huggingface.co/papers",
  "ethics_ai_record_url": "https://ethics.ai/record/18758",
  "original_url": "https://arxiv.org/abs/2608.10636",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}