[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-new-method-speeds-llm-quantization-scales-to-405b-model":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},8577,"new-method-speeds-llm-quantization-scales-to-405b-model","New method speeds LLM quantization, scales to 405B model","A new rotation-based method slashes LLM quantization calibration time and scales to a 405-billion-parameter model on a single GPU.","A new calibration technique makes it dramatically cheaper to shrink big language models down to 4-bit precision.\n\nThinQuant is a new method for rotation learning, the technique that smooths extreme values in a model's activations so it can run at very low bit widths without breaking down. Instead of needing huge amounts of calibration data like prior gradient-free methods such as DartQuant, it picks a much smaller, geometrically representative set of activations and solves a reduced optimization problem with an iterative algorithm. On Llama-3-70B, ThinQuant finished calibration in under 12 minutes and reached a WikiText-2 perplexity of 5.63, compared with 111 minutes and a perplexity of 7.55 for DartQuant. It also scaled to Llama-3.1-405B on a single H200 GPU, a size neither SpinQuant nor DartQuant reportedly manages, finishing calibration in just over two hours with a perplexity of 2.97 versus 3.48 for a rival approach called GPTAQ+QuaRoT.\n\nQuantization is what lets a model trained on a data center cluster run on a single GPU, but the calibration step has been a bottleneck keeping that benefit out of reach for the largest models. Making a 405-billion-parameter model quantizable on one GPU, in hours rather than not at all, turns a research curiosity into something engineers can actually deploy.\n\nLower perplexity scores are a solid signal, not a guarantee the compressed model still reasons or codes as well as the original, so treat these numbers as promising until someone benchmarks it on real tasks.","[\"llm quantization\",\"ai research\",\"model compression\",\"efficient ai\"]","2026-09-30T04:00:00.000Z","2026-09-30T12:32:02.488Z","2026-09-30T12:32:09.218Z","published",null,[],"ai",[26,27,28,29],"llm quantization","ai research","model compression","efficient ai",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2609.36120",0,{"sections":36},[37,40,44,48,53,58,63,68,73,77,82,87,92,97],{"name":38,"slug":24,"count":39,"latest_published_at":18},"AI",5104,{"name":41,"slug":42,"count":43,"latest_published_at":18},"Security","security",785,{"name":45,"slug":46,"count":47,"latest_published_at":18},"Policy","policy",417,{"name":49,"slug":50,"count":51,"latest_published_at":52},"Deals","deals",284,"2026-09-29T21:00:00.000Z",{"name":54,"slug":55,"count":56,"latest_published_at":57},"Hardware","hardware",194,"2026-09-29T13:16:04.000Z",{"name":59,"slug":60,"count":61,"latest_published_at":62},"Science","science",154,"2026-09-28T13:19:18.000Z",{"name":64,"slug":65,"count":66,"latest_published_at":67},"Consumer Tech","consumer-tech",142,"2026-09-29T18:38:03.000Z",{"name":69,"slug":70,"count":71,"latest_published_at":72},"Software","software",91,"2026-09-25T20:55:00.000Z",{"name":74,"slug":75,"count":76,"latest_published_at":18},"Dev Tools","dev-tools",90,{"name":78,"slug":79,"count":80,"latest_published_at":81},"Startups","startups",83,"2026-09-29T21:51:36.000Z",{"name":83,"slug":84,"count":85,"latest_published_at":86},"General","general",49,"2026-09-28T16:44:57.000Z",{"name":88,"slug":89,"count":90,"latest_published_at":91},"Gaming","gaming",48,"2026-09-25T18:35:21.000Z",{"name":93,"slug":94,"count":95,"latest_published_at":96},"Reviews","reviews",31,"2026-09-28T14:31:34.000Z",{"name":98,"slug":99,"count":100,"latest_published_at":101},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]