[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-tokenadapt-swaps-llm-tokenizers-without-costly-retraining":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},8115,"tokenadapt-swaps-llm-tokenizers-without-costly-retraining","TokenAdapt Swaps LLM Tokenizers Without Costly Retraining","A new technique called TokenAdapt lets researchers swap a language model's tokenizer using embedding heuristics instead of expensive full retraining.","Researchers have found a way to give large language models a new vocabulary without burning through a full retraining budget.\n\nA new paper describes TokenAdapt, a method for transplanting one tokenizer onto an already-trained model without the usual exhaustive fine-tuning. It works by estimating embeddings for new tokens two ways: breaking them into subwords using the old tokenizer, and finding semantically similar tokens already in the existing vocabulary. The same paper introduces supertokens, multi-word units learned during pre-tokenization that compress text more efficiently and reduce fragmentation. In testing, the hybrid initialization outperformed existing tokenizer-transplant methods, including TransTokenizer and ReTok, and cut perplexity ratios by at least half compared with ReTok.\n\nTokenizer lock-in is one of those unglamorous problems that quietly limits everything else - it's a big reason models trained mostly on English struggle with other languages or specialized text, and why swapping tokenizers usually means paying for extensive retraining. If TokenAdapt's numbers hold up outside this paper's own benchmarks, it could make multilingual and domain-specific adaptation meaningfully cheaper for teams that don't have retraining budgets to spare.\n\nIt's still a lab result measured against other lab baselines - the real test is whether production model teams actually swap tokenizers instead of just retraining from scratch.","[\"tokenizers\",\"llm-training\",\"ai-research\",\"multilingual-nlp\"]","2026-09-28T04:00:00.000Z","2026-09-28T10:03:35.427Z","2026-09-28T10:03:42.582Z","published",null,[],"ai",[26,27,28,29],"tokenizers","llm-training","ai-research","multilingual-nlp",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2505.09738",0,{"sections":36},[37,40,44,49,54,59,63,68,73,78,83,88,92,97],{"name":38,"slug":24,"count":39,"latest_published_at":18},"AI",4791,{"name":41,"slug":42,"count":43,"latest_published_at":18},"Security","security",762,{"name":45,"slug":46,"count":47,"latest_published_at":48},"Policy","policy",399,"2026-09-27T18:39:02.000Z",{"name":50,"slug":51,"count":52,"latest_published_at":53},"Deals","deals",261,"2026-09-27T15:30:35.000Z",{"name":55,"slug":56,"count":57,"latest_published_at":58},"Hardware","hardware",188,"2026-09-27T20:46:36.000Z",{"name":60,"slug":61,"count":62,"latest_published_at":18},"Science","science",151,{"name":64,"slug":65,"count":66,"latest_published_at":67},"Consumer Tech","consumer-tech",135,"2026-09-26T14:30:00.000Z",{"name":69,"slug":70,"count":71,"latest_published_at":72},"Software","software",91,"2026-09-25T20:55:00.000Z",{"name":74,"slug":75,"count":76,"latest_published_at":77},"Dev Tools","dev-tools",84,"2026-09-26T04:20:58.000Z",{"name":79,"slug":80,"count":81,"latest_published_at":82},"Startups","startups",76,"2026-09-25T18:33:59.000Z",{"name":84,"slug":85,"count":86,"latest_published_at":87},"Gaming","gaming",48,"2026-09-25T18:35:21.000Z",{"name":89,"slug":90,"count":86,"latest_published_at":91},"General","general","2026-09-26T17:02:42.000Z",{"name":93,"slug":94,"count":95,"latest_published_at":96},"Reviews","reviews",30,"2026-09-24T20:07:31.000Z",{"name":98,"slug":99,"count":100,"latest_published_at":101},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]