[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-a-new-transformer-shape-cuts-decoding-time-nearly-in-half":10,"sections":34},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":29,"feedback":33,"feedback_at":22,"cost_usd":33,"total_tokens":33},6370,"a-new-transformer-shape-cuts-decoding-time-nearly-in-half","A New Transformer Shape Cuts Decoding Time Nearly in Half","A new hourglass shaped feedforward design trains faster and speeds up long context decoding without hurting accuracy, researchers report.","Nearly every dense Transformer language model uses the same feed forward shape: narrow, then wide, then narrow again. A new paper asks whether that convention is actually necessary, and the answer looks like no.\n\nThe researchers built what they call Hourglass Transformers, which flip the usual feed forward block into a wide-narrow-wide residual stack and use a matching hourglass attention setup to decouple the model's main hidden-state width from its attention width. That trade lets a model go wider with fewer layers at the same parameter count. Tested across sizes from 113M to 8B parameters, the hourglass models matched conventional Transformers on language modeling and downstream tasks while cutting training compute by 8.7 percent at matched accuracy for the 906M, 3B, and 8B scales. After long context extension, the 8B hourglass model beat its standard counterpart across context lengths from 4k to 64k tokens, and at the 1B scale the smaller attention-layer count delivered up to 1.93 times faster token decoding with half the KV cache memory at 64k context.\n\nThis matters because most public excitement about model architecture goes to bigger context windows or new capabilities, not to the plumbing that decides what inference actually costs. Serving long context is expensive largely because of KV cache size and decode speed, and this work attacks both by changing how width and depth trade off, not by adding tricks on top of the existing shape.\n\nIt is one paper, not a production model, and \"comparable performance\" at smaller scales does not guarantee the same holds at frontier size. Worth watching, not worth rewriting your architecture over yet.","[\"ai\",\"transformer-architecture\",\"model-efficiency\",\"research\"]","2026-09-11T04:00:00.000Z","2026-09-11T09:00:25.550Z","2026-09-11T09:00:37.468Z","published",null,[],"ai",[24,26,27,28],"transformer-architecture","model-efficiency","research",[30],{"name":31,"url":32},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2602.06471",0,{"sections":35},[36,39,43,47,52,57,62,65,70,74,79,84,89,94],{"name":37,"slug":24,"count":38,"latest_published_at":18},"AI",3543,{"name":40,"slug":41,"count":42,"latest_published_at":18},"Security","security",637,{"name":44,"slug":45,"count":46,"latest_published_at":18},"Policy","policy",338,{"name":48,"slug":49,"count":50,"latest_published_at":51},"Deals","deals",179,"2026-06-29T20:02:07.000Z",{"name":53,"slug":54,"count":55,"latest_published_at":56},"Hardware","hardware",153,"2026-09-09T15:12:32.000Z",{"name":58,"slug":59,"count":60,"latest_published_at":61},"Consumer Tech","consumer-tech",99,"2026-09-09T17:27:33.000Z",{"name":63,"slug":64,"count":60,"latest_published_at":18},"Science","science",{"name":66,"slug":67,"count":68,"latest_published_at":69},"Software","software",75,"2026-09-10T20:41:21.000Z",{"name":71,"slug":72,"count":73,"latest_published_at":18},"Dev Tools","dev-tools",70,{"name":75,"slug":76,"count":77,"latest_published_at":78},"Startups","startups",55,"2026-09-09T23:14:29.000Z",{"name":80,"slug":81,"count":82,"latest_published_at":83},"Gaming","gaming",43,"2026-09-10T12:18:06.000Z",{"name":85,"slug":86,"count":87,"latest_published_at":88},"General","general",41,"2026-09-08T01:57:23.000Z",{"name":90,"slug":91,"count":92,"latest_published_at":93},"Reviews","reviews",20,"2026-06-24T12:00:01.000Z",{"name":95,"slug":96,"count":97,"latest_published_at":98},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]