[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-sage-framework-curbs-long-horizon-reasoning-errors-in-llms":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},7785,"sage-framework-curbs-long-horizon-reasoning-errors-in-llms","SAGE Framework Curbs Long Horizon Reasoning Errors in LLMs","A new arXiv paper proposes structural guidance to stop language models from compounding small reasoning errors across long problem chains.","Researchers have a new fix for a stubborn LLM failure mode: reasoning that quietly falls apart the longer it runs.\n\nA paper posted to arXiv describes SAGE (Structural Admissibility-Guided Exploration), a framework aimed at long-horizon reasoning tasks where rewards are sparse and feedback is rare. The authors identify two specific failure patterns: an exploration bias that pulls models toward reasoning branches that look plausible step-by-step but are structurally unstable, and a compounding bias where tiny early deviations snowball across many steps and bury the correct answer. SAGE tackles both with two techniques bolted together: algebraic sparsification, which narrows candidate reasoning steps down to a smaller, well-structured set, and hyperbolic structural guidance, which maps reasoning states into a curved geometric space to give the model steadier signal at every depth. The team tested it across 12 benchmarks and 7 model families.\n\nThe headline result is an up-to-8-fold improvement on the Andrews-Curtis problem, a genuinely hard open math task used as a stress test for long-horizon reasoning. That matters because most LLM reasoning fixes - better prompting, more chain-of-thought, bigger context windows - treat the symptom, not the structural reason models drift off course over many steps. SAGE is instead a bet that the geometry of the reasoning space itself, not just the model doing the reasoning, is where the leverage is.\n\nWhether that generalizes past benchmark tasks to messier real-world agent workflows is the open question, and the code is public for anyone who wants to check.","[\"llm reasoning\",\"arxiv\",\"machine learning research\",\"ai benchmarks\"]","2026-09-25T04:00:00.000Z","2026-09-25T22:11:43.445Z","2026-09-25T22:11:58.797Z","published",null,[],"ai",[26,27,28,29],"llm reasoning","arxiv","machine learning research","ai benchmarks",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2609.30192",0,{"sections":36},[37,41,46,51,56,61,65,70,75,80,85,90,94,99],{"name":38,"slug":24,"count":39,"latest_published_at":40},"AI",4635,"2026-09-26T23:42:06.000Z",{"name":42,"slug":43,"count":44,"latest_published_at":45},"Security","security",751,"2026-09-26T12:00:00.000Z",{"name":47,"slug":48,"count":49,"latest_published_at":50},"Policy","policy",396,"2026-09-26T18:45:15.000Z",{"name":52,"slug":53,"count":54,"latest_published_at":55},"Deals","deals",260,"2026-09-26T09:00:00.000Z",{"name":57,"slug":58,"count":59,"latest_published_at":60},"Hardware","hardware",186,"2026-09-26T17:26:54.000Z",{"name":62,"slug":63,"count":64,"latest_published_at":55},"Science","science",144,{"name":66,"slug":67,"count":68,"latest_published_at":69},"Consumer Tech","consumer-tech",135,"2026-09-26T14:30:00.000Z",{"name":71,"slug":72,"count":73,"latest_published_at":74},"Software","software",91,"2026-09-25T20:55:00.000Z",{"name":76,"slug":77,"count":78,"latest_published_at":79},"Dev Tools","dev-tools",84,"2026-09-26T04:20:58.000Z",{"name":81,"slug":82,"count":83,"latest_published_at":84},"Startups","startups",76,"2026-09-25T18:33:59.000Z",{"name":86,"slug":87,"count":88,"latest_published_at":89},"Gaming","gaming",48,"2026-09-25T18:35:21.000Z",{"name":91,"slug":92,"count":88,"latest_published_at":93},"General","general","2026-09-26T17:02:42.000Z",{"name":95,"slug":96,"count":97,"latest_published_at":98},"Reviews","reviews",30,"2026-09-24T20:07:31.000Z",{"name":100,"slug":101,"count":102,"latest_published_at":103},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]