[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-researchers-tune-llm-sparsity-block-by-block-to-cut-costs":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},9553,"researchers-tune-llm-sparsity-block-by-block-to-cut-costs","Researchers Tune LLM Sparsity Block by Block to Cut Costs","A new technique gives each transformer block its own sparsity budget instead of one fixed rule, squeezing more speed from LLMs without retraining.","A new inference trick lets large language models skip more unneeded math - by deciding, block by block, how much skipping is actually safe.\n\nResearchers describe a training-free method called TopK-Guided that cuts LLM inference costs through activation sparsity, zeroing out activations that barely matter so the model skips computing them. Prior approaches picked one lane: threshold-based methods like TEAL adjust sparsity per token but can't guarantee consistent compute savings, while TopK methods like WINA enforce a fixed sparsity level but apply the same budget to every token and every transformer block, regardless of how sensitive that block is to being pruned. TopK-Guided combines both ideas: it bounds how much sparsity each token gets while allocating bigger sparsity budgets to blocks that can tolerate it and smaller ones to blocks that can't. Tested on Llama-2 and Llama-3 models, it beat both TEAL and WINA on perplexity and downstream accuracy, with the biggest gains at high sparsity, while keeping roughly the same compute overhead as WINA.\n\nThis is the kind of unglamorous efficiency work that actually matters for anyone running LLMs at scale, since it needs no retraining and no new hardware, just a smarter rule for what to skip. Treating transformer blocks as having different sensitivity, instead of applying one global sparsity rule, is a small conceptual shift - but it echoes the same instinct behind other efficiency tricks like mixture-of-experts routing: not every part of a network deserves equal compute.\n\nWhether this earns a spot in production inference stacks depends on real-world latency numbers, not just perplexity charts - the paper does not report wall-clock speedups, so the efficiency claim is still partly on paper.","[\"llm inference\",\"activation sparsity\",\"model efficiency\",\"ai research\"]","2026-10-02T04:00:00.000Z","2026-10-03T00:20:53.451Z","2026-10-03T00:20:59.758Z","published",null,[],"ai",[26,27,28,29],"llm inference","activation sparsity","model efficiency","ai research",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2610.01763",0,{"sections":36},[37,40,44,48,53,57,61,66,71,76,81,86,91,96],{"name":38,"slug":24,"count":39,"latest_published_at":18},"AI",5896,{"name":41,"slug":42,"count":43,"latest_published_at":18},"Security","security",837,{"name":45,"slug":46,"count":47,"latest_published_at":18},"Policy","policy",438,{"name":49,"slug":50,"count":51,"latest_published_at":52},"Deals","deals",317,"2026-10-01T22:00:00.000Z",{"name":54,"slug":55,"count":56,"latest_published_at":18},"Hardware","hardware",199,{"name":58,"slug":59,"count":60,"latest_published_at":18},"Science","science",171,{"name":62,"slug":63,"count":64,"latest_published_at":65},"Consumer Tech","consumer-tech",155,"2026-10-01T19:54:10.000Z",{"name":67,"slug":68,"count":69,"latest_published_at":70},"Dev Tools","dev-tools",96,"2026-10-01T16:57:03.000Z",{"name":72,"slug":73,"count":74,"latest_published_at":75},"Software","software",93,"2026-09-30T21:41:11.000Z",{"name":77,"slug":78,"count":79,"latest_published_at":80},"Startups","startups",90,"2026-10-01T21:55:22.000Z",{"name":82,"slug":83,"count":84,"latest_published_at":85},"Gaming","gaming",53,"2026-10-02T02:50:39.000Z",{"name":87,"slug":88,"count":89,"latest_published_at":90},"General","general",50,"2026-09-30T21:37:54.000Z",{"name":92,"slug":93,"count":94,"latest_published_at":95},"Reviews","reviews",31,"2026-09-28T14:31:34.000Z",{"name":97,"slug":98,"count":99,"latest_published_at":100},"How-To","how-to",7,"2026-10-01T09:00:00.000Z"]