[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-researchers-cut-ai-reasoning-costs-with-a-582-parameter-controller":10,"sections":34},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":29,"feedback":33,"feedback_at":22,"cost_usd":33,"total_tokens":33},8303,"researchers-cut-ai-reasoning-costs-with-a-582-parameter-controller","Researchers Cut AI Reasoning Costs With a 582 Parameter Controller","A tiny controller reads a model's internal reasoning state to skip wasted steps, cutting tokens by over 60 percent without a bigger model or longer search.","A new paper proposes letting language models decide their own reasoning path, instead of following a fixed script.\n\nResearchers behind a paper posted to arXiv describe State of Thought (SoT), a technique that extracts a compact state from a model's internal information flow and feeds it to a small 582-parameter controller. That controller runs on top of a frozen, unmodified backbone model and decides which pieces of prior reasoning are worth reusing for the current problem, rather than forcing the model to regenerate a full chain of thought or search through a fixed tree of options. Tested on three LLMs across 16 datasets, SoT improved accuracy on quantitative, general, symbolic-and-code, and long-context reasoning tasks - gains reached 2.51x on long-context problems - while cutting generated tokens by 62.6% and latency by 44.6%. Applied to vision-language models, it added 3.8 accuracy points while using 74.9% fewer completion tokens and 73.5% less time than search-based approaches.\n\nMost test-time reasoning gains right now come from brute force - longer chains of thought, wider search trees, more sampled paths - which is slow and expensive to run at scale. A 582-parameter add-on that improves accuracy while cutting both token usage and latency, without retraining the underlying model, suggests reasoning quality and reasoning cost are less tightly coupled than the industry has assumed. That's a meaningful data point for anyone paying per-token API bills.\n\nThe numbers come from the paper's own benchmarks, not independent replication, so treat the efficiency claims as a strong opening bid rather than settled science.","[\"ai\",\"large-language-models\",\"reasoning\",\"research\"]","2026-09-28T04:00:00.000Z","2026-09-28T22:48:34.182Z","2026-09-28T22:48:40.063Z","published",null,[],"ai",[24,26,27,28],"large-language-models","reasoning","research",[30],{"name":31,"url":32},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2609.16055",0,{"sections":35},[36,40,45,50,55,60,65,70,75,80,85,90,95,100],{"name":37,"slug":24,"count":38,"latest_published_at":39},"AI",4900,"2026-09-28T17:44:43.000Z",{"name":41,"slug":42,"count":43,"latest_published_at":44},"Security","security",766,"2026-09-28T15:35:23.000Z",{"name":46,"slug":47,"count":48,"latest_published_at":49},"Policy","policy",405,"2026-09-28T17:00:51.000Z",{"name":51,"slug":52,"count":53,"latest_published_at":54},"Deals","deals",269,"2026-09-28T17:42:46.000Z",{"name":56,"slug":57,"count":58,"latest_published_at":59},"Hardware","hardware",191,"2026-09-28T15:45:00.000Z",{"name":61,"slug":62,"count":63,"latest_published_at":64},"Science","science",154,"2026-09-28T13:19:18.000Z",{"name":66,"slug":67,"count":68,"latest_published_at":69},"Consumer Tech","consumer-tech",139,"2026-09-28T17:09:47.000Z",{"name":71,"slug":72,"count":73,"latest_published_at":74},"Software","software",91,"2026-09-25T20:55:00.000Z",{"name":76,"slug":77,"count":78,"latest_published_at":79},"Dev Tools","dev-tools",87,"2026-09-28T16:11:42.000Z",{"name":81,"slug":82,"count":83,"latest_published_at":84},"Startups","startups",80,"2026-09-28T17:50:28.000Z",{"name":86,"slug":87,"count":88,"latest_published_at":89},"General","general",49,"2026-09-28T16:44:57.000Z",{"name":91,"slug":92,"count":93,"latest_published_at":94},"Gaming","gaming",48,"2026-09-25T18:35:21.000Z",{"name":96,"slug":97,"count":98,"latest_published_at":99},"Reviews","reviews",31,"2026-09-28T14:31:34.000Z",{"name":101,"slug":102,"count":103,"latest_published_at":104},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]