{
  "id": 11590,
  "url": "https://arxiv.org/abs/2607.15232v1",
  "title": "In-Place Tokenizer Expansion for Pre-trained LLMs",
  "summary": "A tokenizer fixed at the start of pre-training allocates vocabulary in proportion to the pre-training corpus, reflecting the deployment priorities at that time. When those priorities shift, languages added later are split into many more tokens per word, which can raise latency, compute, and energy consumption for users of those languages. Cloud models can afford a broad vocabulary because the embedding and LM-head matrices are a small fraction of their parameters. On a compact model those matric",
  "authors": "Jimmy T. H. Smith, Tarek Dakhran, Alberto Cabrera, Simon S. Lee, Paul Pak, Aditya Tadimeti, Tim Seyde, Maxime Labonne, Alexander Amini, Mathias Lechner",
  "category": "research",
  "topics": "environment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-07-16T17:32:38.000Z",
  "fetched_at": "2026-07-18T05:10:55.931Z",
  "source_slug": "x-arxiv-cs-ai",
  "source_name": "arXiv cs.AI",
  "source_homepage": "https://arxiv.org/list/cs.AI/recent",
  "ethics_ai_record_url": "https://ethics.ai/record/11590",
  "original_url": "https://arxiv.org/abs/2607.15232v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}