[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-ai-agents-get-a-fix-for-tunnel-vision-on-decision-trees":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},6349,"ai-agents-get-a-fix-for-tunnel-vision-on-decision-trees","AI Agents Get a Fix for Tunnel Vision on Decision Trees","A new offline training method called DDO helps AI agents remember multiple winning strategies instead of collapsing onto just one.","Researchers have a new way to stop AI agents from forgetting their backup plans.\n\nMost LLM agents trained for multi-step tasks get graded only on whether the final outcome succeeded. That single pass-fail signal tells the model nothing about the other winning paths it could have taken from the same decision point. A new paper introduces Direct Diversity Optimization (DDO), an offline post-training method built from two pieces: Divergence-Tree Collection, which maps out branch sets rooted at shared decision states, and a Reference-Relative Target-Odds Objective, which trains the model to weigh successful alternatives against each other rather than just chasing the one path it already knows works. Tested on three benchmark environments, BabyAI, BabaIsAI, and WebShop, DDO beat comparison methods on both raw task success and on how many distinct successful strategies the model could still produce under a fixed rollout budget. It also recovered better after researchers swapped out a local action mid-trajectory, a rough proxy for handling a plan getting disrupted partway through.\n\nThis matters because narrow strategy coverage is a quiet failure mode in agent training. A model that only knows one route to success looks fine on a benchmark leaderboard right up until that one route is blocked, whether by a UI change, an unexpected error, or a competitor's environment quirk. Standard outcome-only post-training and successful-only imitation both tend to collapse a model onto whichever path was reinforced first, which is efficient for the benchmark but brittle in the messier real world agents actually operate in.\n\nIt is a small-scale academic result on toy and semi-realistic environments, not a production agent framework, so treat the specific numbers as early signal rather than a verdict on how commercial coding or shopping agents will behave at scale.","[\"ai-agents\",\"llm-training\",\"reinforcement-learning\",\"arxiv\"]","2026-09-11T04:00:00.000Z","2026-09-11T07:41:31.397Z","2026-09-11T07:41:43.408Z","published",null,[],"ai",[26,27,28,29],"ai-agents","llm-training","reinforcement-learning","arxiv",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2609.10052",0,{"sections":36},[37,40,44,48,53,58,63,66,71,75,80,85,90,95],{"name":38,"slug":24,"count":39,"latest_published_at":18},"AI",3522,{"name":41,"slug":42,"count":43,"latest_published_at":18},"Security","security",637,{"name":45,"slug":46,"count":47,"latest_published_at":18},"Policy","policy",338,{"name":49,"slug":50,"count":51,"latest_published_at":52},"Deals","deals",179,"2026-06-29T20:02:07.000Z",{"name":54,"slug":55,"count":56,"latest_published_at":57},"Hardware","hardware",153,"2026-09-09T15:12:32.000Z",{"name":59,"slug":60,"count":61,"latest_published_at":62},"Consumer Tech","consumer-tech",99,"2026-09-09T17:27:33.000Z",{"name":64,"slug":65,"count":61,"latest_published_at":18},"Science","science",{"name":67,"slug":68,"count":69,"latest_published_at":70},"Software","software",75,"2026-09-10T20:41:21.000Z",{"name":72,"slug":73,"count":74,"latest_published_at":18},"Dev Tools","dev-tools",70,{"name":76,"slug":77,"count":78,"latest_published_at":79},"Startups","startups",55,"2026-09-09T23:14:29.000Z",{"name":81,"slug":82,"count":83,"latest_published_at":84},"Gaming","gaming",43,"2026-09-10T12:18:06.000Z",{"name":86,"slug":87,"count":88,"latest_published_at":89},"General","general",41,"2026-09-08T01:57:23.000Z",{"name":91,"slug":92,"count":93,"latest_published_at":94},"Reviews","reviews",20,"2026-06-24T12:00:01.000Z",{"name":96,"slug":97,"count":98,"latest_published_at":99},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]