[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-a-sparser-way-to-shrink-trillion-parameter-ai-models":10,"sections":34},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":29,"feedback":33,"feedback_at":22,"cost_usd":33,"total_tokens":33},9966,"a-sparser-way-to-shrink-trillion-parameter-ai-models","A Sparser Way to Shrink Trillion-Parameter AI Models","A new hardware-software framework compresses trillion-parameter MoE models with sparse quantization, cutting latency up to 4x on B200 GPUs.","Researchers have found a way to shrink trillion-parameter AI models without gutting their performance.\n\nA new paper describes a hardware-software framework that compresses the expert weights inside Mixture-of-Experts (MoE) models - the architecture behind today's largest language models - into a low-precision, sparse format built for Nvidia's Sparse Tensor Cores. The trick is a differentiable optimization method that jointly tunes sparsity patterns and quantization instead of treating them as separate steps, plus a custom sparse matrix-multiplication kernel for MoE inference. Tested on models from 30 billion to 1 trillion parameters, the approach lifts state-of-the-art compression accuracy by up to 4.35 percentage points while keeping 96.09% of the original model's performance. On Nvidia B200 GPUs, the custom kernel beat Nvidia's own baseline by up to 1.65x, improving serving throughput by 1.18x and cutting end-to-end latency by up to 4.03x.\n\nThis matters because MoE scaling has a dirty secret: the bigger the expert pool, the more memory bandwidth gets burned just moving weights around, not computing with them. Compression methods that hurt accuracy or ignore the actual hardware have been a stopgap at best. Pairing the algorithm directly to sparse tensor-core hardware - rather than bolting compression on after the fact - is the kind of co-design that turns a lab result into something a cloud provider could actually ship.\n\nNone of this changes what these models can do. It just makes running trillion-parameter models cheaper and faster, which is where the real deployment bottleneck has been all along.","[\"ai\",\"mixture-of-experts\",\"gpu-hardware\",\"model-compression\"]","2026-10-05T04:00:00.000Z","2026-10-05T15:43:00.986Z","2026-10-05T15:43:06.783Z","published",null,[],"ai",[24,26,27,28],"mixture-of-experts","gpu-hardware","model-compression",[30],{"name":31,"url":32},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2610.02241",0,{"sections":35},[36,39,43,48,53,58,62,67,71,75,80,85,90,95],{"name":37,"slug":24,"count":38,"latest_published_at":18},"AI",6233,{"name":40,"slug":41,"count":42,"latest_published_at":18},"Security","security",868,{"name":44,"slug":45,"count":46,"latest_published_at":47},"Policy","policy",444,"2026-10-03T15:02:01.000Z",{"name":49,"slug":50,"count":51,"latest_published_at":52},"Deals","deals",323,"2026-10-04T13:00:00.000Z",{"name":54,"slug":55,"count":56,"latest_published_at":57},"Hardware","hardware",204,"2026-10-03T14:50:50.000Z",{"name":59,"slug":60,"count":61,"latest_published_at":18},"Science","science",177,{"name":63,"slug":64,"count":65,"latest_published_at":66},"Consumer Tech","consumer-tech",158,"2026-10-03T03:21:12.000Z",{"name":68,"slug":69,"count":70,"latest_published_at":18},"Dev Tools","dev-tools",97,{"name":72,"slug":73,"count":70,"latest_published_at":74},"Software","software","2026-10-04T10:00:00.000Z",{"name":76,"slug":77,"count":78,"latest_published_at":79},"Startups","startups",92,"2026-10-04T14:36:25.000Z",{"name":81,"slug":82,"count":83,"latest_published_at":84},"Gaming","gaming",53,"2026-10-02T02:50:39.000Z",{"name":86,"slug":87,"count":88,"latest_published_at":89},"General","general",51,"2026-10-05T02:35:01.000Z",{"name":91,"slug":92,"count":93,"latest_published_at":94},"Reviews","reviews",32,"2026-10-02T18:00:00.000Z",{"name":96,"slug":97,"count":98,"latest_published_at":99},"How-To","how-to",7,"2026-10-01T09:00:00.000Z"]