[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-new-framework-compares-ai-puzzle-solving-across-three-axes":10,"sections":34},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":29,"feedback":33,"feedback_at":22,"cost_usd":33,"total_tokens":33},10889,"new-framework-compares-ai-puzzle-solving-across-three-axes","New Framework Compares AI Puzzle Solving Across Three Axes","A three-axis framework compares graph search, reinforcement learning, and LLM-based reasoning on Tower of Hanoi, showing LLMs cost far more compute.","A new benchmarking framework puts three very different flavors of AI puzzle-solving on the same scorecard, and the LLMs don't look great on cost.\n\nResearchers built a three-dimensional framework that maps AI decision-making methods onto the Markov decision process (MDP) formalism, ranks how much human-designed structure each method leans on, and measures the skill and computational cost each one burns through. They tested it on the Tower of Hanoi, a puzzle with simple rules but complexity that scales fast as you add disks. Four approaches went head to head: Neurosolver, a graph-based solver; forward-backward reinforcement learning (FBRL); automated thought-of-search (AutoToS), an LLM-based method; and a two-agent variant called DA-ToS. The framework let the team compare all of them under the same conditions instead of relying on separate papers' cherry-picked benchmarks.\n\nThe standout finding is that LLM-based solvers don't actually encode less human knowledge than the alternatives - they just move the work from architecture design to inference-time verification, checking and rechecking candidate moves as they go. That shift makes them substantially more expensive in memory and runtime than the graph-based and reinforcement-learning methods, even when solving the identical puzzle.\n\nTranslation: making a model sound like it's reasoning through a puzzle costs real compute, and that style of reasoning still can't out-hustle a well-tuned search algorithm on its own turf.","[\"ai\",\"llm-reasoning\",\"reinforcement-learning\",\"benchmarks\"]","2026-10-09T04:00:00.000Z","2026-10-09T20:12:08.240Z","2026-10-09T20:12:12.093Z","published",null,[],"ai",[24,26,27,28],"llm-reasoning","reinforcement-learning","benchmarks",[30],{"name":31,"url":32},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2610.11696",0,{"sections":35},[36,39,43,48,53,58,62,67,72,77,82,87,92,97],{"name":37,"slug":24,"count":38,"latest_published_at":18},"AI",6619,{"name":40,"slug":41,"count":42,"latest_published_at":18},"Security","security",926,{"name":44,"slug":45,"count":46,"latest_published_at":47},"Policy","policy",486,"2026-10-08T22:40:11.000Z",{"name":49,"slug":50,"count":51,"latest_published_at":52},"Deals","deals",474,"2026-10-08T22:00:00.000Z",{"name":54,"slug":55,"count":56,"latest_published_at":57},"Hardware","hardware",229,"2026-10-08T20:47:10.000Z",{"name":59,"slug":60,"count":61,"latest_published_at":18},"Science","science",192,{"name":63,"slug":64,"count":65,"latest_published_at":66},"Consumer Tech","consumer-tech",181,"2026-10-08T23:26:35.000Z",{"name":68,"slug":69,"count":70,"latest_published_at":71},"Startups","startups",117,"2026-10-08T16:45:00.000Z",{"name":73,"slug":74,"count":75,"latest_published_at":76},"Software","software",114,"2026-10-08T17:57:01.000Z",{"name":78,"slug":79,"count":80,"latest_published_at":81},"Dev Tools","dev-tools",105,"2026-10-07T16:59:11.000Z",{"name":83,"slug":84,"count":85,"latest_published_at":86},"General","general",66,"2026-10-09T04:46:11.000Z",{"name":88,"slug":89,"count":90,"latest_published_at":91},"Gaming","gaming",58,"2026-10-08T20:08:45.000Z",{"name":93,"slug":94,"count":95,"latest_published_at":96},"Reviews","reviews",34,"2026-10-08T14:00:22.000Z",{"name":98,"slug":99,"count":100,"latest_published_at":101},"How-To","how-to",8,"2026-10-05T09:00:00.000Z"]