[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-study-finds-llms-use-a-two-phase-computation-pattern":10,"sections":34},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":29,"feedback":33,"feedback_at":22,"cost_usd":33,"total_tokens":33},6377,"study-finds-llms-use-a-two-phase-computation-pattern","Study Finds LLMs Use a Two Phase Computation Pattern","A new tracing method shows large language models compute a rough answer early, then refine it with denser attention in later layers.","Not every part of a large language model fires for every prompt, and a new paper shows just how little of one actually gets used per query.\n\nResearchers built a technique called s-Trace that estimates a small subgraph of a model's full computational graph, one that still approximates the model's real output. Running it across several LLMs, they found computation splits into two phases. A sparse set of early-layer nodes produces a rough first guess, essentially the most likely answer, while later layers, increasingly made up of attention heads, refine that guess into the full probability distribution. How much extra computation a query needs tracks with how uncertain the model is about its answer. The leanest subgraphs mostly encode shallow statistics, like how often a word shows up, rather than anything resembling reasoning.\n\nThat's evidence LLMs have an informal internal division of labor, a cheap first pass followed by optional deeper refinement, even though nobody designed them to work that way. It matters for anyone trying to cut inference costs: if you can tell which inputs only need the early-layer core, you could skip the expensive later layers on easy queries without touching accuracy.\n\nWorth remembering this describes existing behavior, it is not a new architecture, so any actual speedup still requires someone to engineer a system around the finding.","[\"ai\",\"llms\",\"interpretability\",\"research\"]","2026-09-11T04:00:00.000Z","2026-09-11T09:22:44.847Z","2026-09-11T09:22:56.757Z","published",null,[],"ai",[24,26,27,28],"llms","interpretability","research",[30],{"name":31,"url":32},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2605.27033",0,{"sections":35},[36,39,43,47,52,57,62,65,70,74,79,84,89,94],{"name":37,"slug":24,"count":38,"latest_published_at":18},"AI",3543,{"name":40,"slug":41,"count":42,"latest_published_at":18},"Security","security",637,{"name":44,"slug":45,"count":46,"latest_published_at":18},"Policy","policy",338,{"name":48,"slug":49,"count":50,"latest_published_at":51},"Deals","deals",179,"2026-06-29T20:02:07.000Z",{"name":53,"slug":54,"count":55,"latest_published_at":56},"Hardware","hardware",153,"2026-09-09T15:12:32.000Z",{"name":58,"slug":59,"count":60,"latest_published_at":61},"Consumer Tech","consumer-tech",99,"2026-09-09T17:27:33.000Z",{"name":63,"slug":64,"count":60,"latest_published_at":18},"Science","science",{"name":66,"slug":67,"count":68,"latest_published_at":69},"Software","software",75,"2026-09-10T20:41:21.000Z",{"name":71,"slug":72,"count":73,"latest_published_at":18},"Dev Tools","dev-tools",70,{"name":75,"slug":76,"count":77,"latest_published_at":78},"Startups","startups",55,"2026-09-09T23:14:29.000Z",{"name":80,"slug":81,"count":82,"latest_published_at":83},"Gaming","gaming",43,"2026-09-10T12:18:06.000Z",{"name":85,"slug":86,"count":87,"latest_published_at":88},"General","general",41,"2026-09-08T01:57:23.000Z",{"name":90,"slug":91,"count":92,"latest_published_at":93},"Reviews","reviews",20,"2026-06-24T12:00:01.000Z",{"name":95,"slug":96,"count":97,"latest_published_at":98},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]