[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-maple-reshuffles-moe-experts-beats-full-model-at-75":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},5362,"maple-reshuffles-moe-experts-beats-full-model-at-75","MAPLE Reshuffles MoE Experts, Beats Full Model at 75%","A new allocation method lets MoE models drop a quarter of their experts yet beat full-capacity accuracy while cutting serving latency by a third.","Researchers have found that mixture-of-experts models don't need the same number of active experts in every layer, and skipping that assumption makes them faster and more accurate.\n\nMost sparsely-activated MoE Transformers route a fixed number of experts per layer, regardless of how much each layer actually needs. A new framework called MAPLE (MoE Adaptive Plug-and-play Layer-wise Expert allocation) probes each layer's sensitivity to expert count, then uses a closed-form calculation plus a genetic search to reassign the expert budget: more experts for sensitive layers, fewer for redundant ones. It requires no retraining and no weight changes, applying after the fact to an existing pretrained model. Tested on four MoE models of different scales and architectures, MAPLE beat both uniform allocation and pruning-based baselines while using only 75% of the routed-expert budget.\n\nOn DeepSeek-MoE-16B, a MAPLE-allocated version running on just 75% of experts scored higher than the original full-capacity baseline on three benchmarks (ARC-E, ARC-C, and BoolQ), and that accuracy bump came with a 32.2% latency cut and a 47.4% throughput gain in real serving tests using SGLang. That is a rare case where cutting compute improves both cost and quality at once, rather than trading one for the other.\n\nMoE efficiency work usually means either pruning experts wholesale or activating more of them per token; MAPLE's bet is that the fixed per-layer expert count was the wrong assumption all along, and these numbers suggest it was right, though it is still one paper's benchmarks, not a production deployment.","[\"mixture-of-experts\",\"llm-inference\",\"model-efficiency\",\"research\"]","2026-08-18T04:00:00.000Z","2026-08-18T16:59:58.470Z","2026-08-18T17:00:10.348Z","published",null,[],"ai",[26,27,28,29],"mixture-of-experts","llm-inference","model-efficiency","research",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2608.15299",0,{"sections":36},[37,41,45,50,55,60,65,70,75,79,84,89,94,99],{"name":38,"slug":24,"count":39,"latest_published_at":40},"AI",3293,"2026-08-20T04:00:00.000Z",{"name":42,"slug":43,"count":44,"latest_published_at":40},"Security","security",435,{"name":46,"slug":47,"count":48,"latest_published_at":49},"Policy","policy",210,"2026-08-19T09:32:27.000Z",{"name":51,"slug":52,"count":53,"latest_published_at":54},"Deals","deals",179,"2026-06-29T20:02:07.000Z",{"name":56,"slug":57,"count":58,"latest_published_at":59},"Hardware","hardware",140,"2026-08-19T18:25:42.000Z",{"name":61,"slug":62,"count":63,"latest_published_at":64},"Consumer Tech","consumer-tech",95,"2026-08-18T16:05:00.000Z",{"name":66,"slug":67,"count":68,"latest_published_at":69},"Science","science",90,"2026-08-19T18:41:02.000Z",{"name":71,"slug":72,"count":73,"latest_published_at":74},"Software","software",73,"2026-08-18T07:51:50.000Z",{"name":76,"slug":77,"count":78,"latest_published_at":18},"Dev Tools","dev-tools",69,{"name":80,"slug":81,"count":82,"latest_published_at":83},"Startups","startups",47,"2026-08-19T19:13:46.000Z",{"name":85,"slug":86,"count":87,"latest_published_at":88},"Gaming","gaming",41,"2026-07-09T04:00:00.000Z",{"name":90,"slug":91,"count":92,"latest_published_at":93},"General","general",33,"2026-08-18T22:18:13.000Z",{"name":95,"slug":96,"count":97,"latest_published_at":98},"Reviews","reviews",20,"2026-06-24T12:00:01.000Z",{"name":100,"slug":101,"count":102,"latest_published_at":103},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]