[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-new-transformer-design-ditches-global-attention-for-speed":10,"sections":34},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":29,"feedback":33,"feedback_at":22,"cost_usd":33,"total_tokens":33},5048,"new-transformer-design-ditches-global-attention-for-speed","New Transformer Design Ditches Global Attention for Speed","BCMT swaps dense global attention for block-level processing plus causal memory, matching Transformer accuracy while training faster and using less memory.","A new transformer architecture ditches global attention for local blocks and a compressed memory trail.\n\nResearchers have introduced BCMT (Blockwise Causal Memory Transformer), a language model architecture built to sidestep the quadratic cost of standard self-attention. Instead of comparing every token to every other token, BCMT runs dense attention only within small local blocks. Each block then produces a summary, and those summaries get folded into an exponential causal memory that flows forward into later blocks. In tests on context lengths up to 1024 tokens, BCMT matched the validation performance of standard dense transformers while training faster and using less memory, and an ablation study traced those gains specifically to the memory mechanism.\n\nThis targets a real bottleneck. Attention's quadratic scaling is the reason long-context models are expensive to train and run, and it's why linear attention, state-space models, and recurrent memory schemes keep showing up in the research literature. BCMT's pitch is compatibility: it slots into existing dense-attention implementations rather than replacing them outright, which matters more for adoption than the underlying math.\n\nThe catch is scale. 1024 tokens is a modest context window by 2026 standards, when production models routinely advertise context windows in the hundreds of thousands. Whether this memory trick holds up, and stays cheap, at that length is the test that actually matters.","[\"ai\",\"transformers\",\"long-context\",\"efficiency\"]","2026-08-17T04:00:00.000Z","2026-08-17T07:18:18.320Z","2026-08-17T07:18:30.137Z","published",null,[],"ai",[24,26,27,28],"transformers","long-context","efficiency",[30],{"name":31,"url":32},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2608.13578",0,{"sections":35},[36,40,44,49,54,59,64,69,74,79,84,89,94,99],{"name":37,"slug":24,"count":38,"latest_published_at":39},"AI",3293,"2026-08-20T04:00:00.000Z",{"name":41,"slug":42,"count":43,"latest_published_at":39},"Security","security",435,{"name":45,"slug":46,"count":47,"latest_published_at":48},"Policy","policy",210,"2026-08-19T09:32:27.000Z",{"name":50,"slug":51,"count":52,"latest_published_at":53},"Deals","deals",179,"2026-06-29T20:02:07.000Z",{"name":55,"slug":56,"count":57,"latest_published_at":58},"Hardware","hardware",140,"2026-08-19T18:25:42.000Z",{"name":60,"slug":61,"count":62,"latest_published_at":63},"Consumer Tech","consumer-tech",95,"2026-08-18T16:05:00.000Z",{"name":65,"slug":66,"count":67,"latest_published_at":68},"Science","science",90,"2026-08-19T18:41:02.000Z",{"name":70,"slug":71,"count":72,"latest_published_at":73},"Software","software",73,"2026-08-18T07:51:50.000Z",{"name":75,"slug":76,"count":77,"latest_published_at":78},"Dev Tools","dev-tools",69,"2026-08-18T04:00:00.000Z",{"name":80,"slug":81,"count":82,"latest_published_at":83},"Startups","startups",47,"2026-08-19T19:13:46.000Z",{"name":85,"slug":86,"count":87,"latest_published_at":88},"Gaming","gaming",41,"2026-07-09T04:00:00.000Z",{"name":90,"slug":91,"count":92,"latest_published_at":93},"General","general",33,"2026-08-18T22:18:13.000Z",{"name":95,"slug":96,"count":97,"latest_published_at":98},"Reviews","reviews",20,"2026-06-24T12:00:01.000Z",{"name":100,"slug":101,"count":102,"latest_published_at":103},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]