[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-researchers-map-which-ai-expert-layers-are-safe-to-cut":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},5002,"researchers-map-which-ai-expert-layers-are-safe-to-cut","Researchers Map Which AI Expert Layers Are Safe to Cut","A depth-aware study of Mixture-of-Experts models finds late layers tolerate aggressive expert pruning far better than early ones.","A new study on arXiv pinpoints exactly which layers of a giant AI model can be trimmed without breaking it.\n\nResearchers ran a depth-aware sensitivity analysis on Qwen3.6-35B-A3B, a Mixture-of-Experts model with 40 layers and 256 experts per layer, by masking low-magnitude experts and checking output quality on a cross-lingual code-translation benchmark. They tested at three scales - 100, 300, and 500 prompts - across three H100 GPU servers. A flat 30% masking rate applied to every layer wrecked quality, keeping only 150 of 300 \"good or similar\" outputs. But masking concentrated in the last few layers, especially layers 35 through 39, held up far better: one late-focused policy kept 419 of 500 acceptable outputs on a held-out validation set while removing just 640 of the model's 10,240 total experts.\n\nThe finding backs up something MoE researchers have long suspected but rarely measured directly: early and middle layers do the structural heavy lifting, while late layers carry more redundant, safely prunable capacity. That distinction turns expert pruning from guesswork into an actual policy - a depth-aware map instead of blanket cuts. For anyone footing the GPU bill to serve these sprawling models, a validated pruning target beats trial and error.\n\nThe team also cut routing width from 8 active experts per token to 6, getting a real speed boost with no quality loss - though it did not yet combine cleanly with the layer-masking approach. Efficiency tricks, it turns out, don't always stack.","[\"mixture-of-experts\",\"model compression\",\"llm efficiency\",\"ai research\"]","2026-08-17T04:00:00.000Z","2026-08-17T04:54:32.740Z","2026-08-17T04:54:44.583Z","published",null,[],"ai",[26,27,28,29],"mixture-of-experts","model compression","llm efficiency","ai research",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2608.13565",0,{"sections":36},[37,41,45,50,55,60,65,70,75,80,85,90,95,100],{"name":38,"slug":24,"count":39,"latest_published_at":40},"AI",3293,"2026-08-20T04:00:00.000Z",{"name":42,"slug":43,"count":44,"latest_published_at":40},"Security","security",435,{"name":46,"slug":47,"count":48,"latest_published_at":49},"Policy","policy",210,"2026-08-19T09:32:27.000Z",{"name":51,"slug":52,"count":53,"latest_published_at":54},"Deals","deals",179,"2026-06-29T20:02:07.000Z",{"name":56,"slug":57,"count":58,"latest_published_at":59},"Hardware","hardware",140,"2026-08-19T18:25:42.000Z",{"name":61,"slug":62,"count":63,"latest_published_at":64},"Consumer Tech","consumer-tech",95,"2026-08-18T16:05:00.000Z",{"name":66,"slug":67,"count":68,"latest_published_at":69},"Science","science",90,"2026-08-19T18:41:02.000Z",{"name":71,"slug":72,"count":73,"latest_published_at":74},"Software","software",73,"2026-08-18T07:51:50.000Z",{"name":76,"slug":77,"count":78,"latest_published_at":79},"Dev Tools","dev-tools",69,"2026-08-18T04:00:00.000Z",{"name":81,"slug":82,"count":83,"latest_published_at":84},"Startups","startups",47,"2026-08-19T19:13:46.000Z",{"name":86,"slug":87,"count":88,"latest_published_at":89},"Gaming","gaming",41,"2026-07-09T04:00:00.000Z",{"name":91,"slug":92,"count":93,"latest_published_at":94},"General","general",33,"2026-08-18T22:18:13.000Z",{"name":96,"slug":97,"count":98,"latest_published_at":99},"Reviews","reviews",20,"2026-06-24T12:00:01.000Z",{"name":101,"slug":102,"count":103,"latest_published_at":104},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]