[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-new-rl-algorithm-aims-to-fix-ppos-stability-tradeoff":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},7580,"new-rl-algorithm-aims-to-fix-ppos-stability-tradeoff","New RL Algorithm Aims to Fix PPO's Stability Tradeoff","ANO, a new policy-optimization algorithm, beats PPO and SPO on Atari and MuJoCo benchmarks and barely degrades under aggressive learning rates.","A new algorithm called ANO claims to fix a stability problem that has long dogged PPO (Proximal Policy Optimization), the default method for training AI agents and aligning large language models.\n\nResearchers built Anchored Neighborhood Optimization (ANO) to replace the blunt mechanisms other algorithms use to keep policy updates in check. PPO relies on hard clipping: once an update strays too far from the current policy, it simply stops correcting and lets the policy drift unsupervised. A rival method called SPO closes that gap with a penalty that grows without limit, which controls drift but can destabilize training at higher learning rates. ANO replaces both approaches with a smooth, bounded correction that pulls extreme updates back toward the trust region without ever switching off entirely or growing without bound.\n\nAcross 40 Atari games and the MuJoCo physics benchmarks, ANO topped the field on both scoring metrics the paper tracks, something none of its individual rivals managed in both environments. The gap shows up most when training goes wrong: raising the learning rate roughly threefold collapsed PPO's performance by 54.5%, while ANO's score dropped by less than 1%. Since fine-tuning AI agents and aligning chatbots with human feedback both live or die on learning-rate choices, an algorithm that shrugs off a bad setting could mean fewer wasted training runs.\n\nStill, these are Atari and MuJoCo results, not a large-scale LLM alignment run. That's the setting where PPO's brittleness actually costs real money, and where ANO hasn't been tested yet.","[\"reinforcement-learning\",\"ppo\",\"algorithms\",\"ai-research\"]","2026-09-24T04:00:00.000Z","2026-09-24T07:43:56.580Z","2026-09-24T07:44:01.382Z","published",null,[],"ai",[26,27,28,29],"reinforcement-learning","ppo","algorithms","ai-research",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2605.02320",0,{"sections":36},[37,40,44,49,54,59,64,69,74,79,84,89,94,99],{"name":38,"slug":24,"count":39,"latest_published_at":18},"AI",4424,{"name":41,"slug":42,"count":43,"latest_published_at":18},"Security","security",724,{"name":45,"slug":46,"count":47,"latest_published_at":48},"Policy","policy",380,"2026-09-23T22:53:43.000Z",{"name":50,"slug":51,"count":52,"latest_published_at":53},"Deals","deals",227,"2026-09-24T11:08:33.000Z",{"name":55,"slug":56,"count":57,"latest_published_at":58},"Hardware","hardware",174,"2026-09-24T10:10:29.000Z",{"name":60,"slug":61,"count":62,"latest_published_at":63},"Science","science",136,"2026-09-24T09:00:00.000Z",{"name":65,"slug":66,"count":67,"latest_published_at":68},"Consumer Tech","consumer-tech",116,"2026-09-24T00:51:49.000Z",{"name":70,"slug":71,"count":72,"latest_published_at":73},"Software","software",85,"2026-09-23T20:00:00.000Z",{"name":75,"slug":76,"count":77,"latest_published_at":78},"Dev Tools","dev-tools",79,"2026-09-22T22:21:13.000Z",{"name":80,"slug":81,"count":82,"latest_published_at":83},"Startups","startups",66,"2026-09-23T17:28:38.000Z",{"name":85,"slug":86,"count":87,"latest_published_at":88},"Gaming","gaming",45,"2026-09-22T15:35:06.000Z",{"name":90,"slug":91,"count":92,"latest_published_at":93},"General","general",43,"2026-09-21T23:48:56.000Z",{"name":95,"slug":96,"count":97,"latest_published_at":98},"Reviews","reviews",27,"2026-09-22T13:00:00.000Z",{"name":100,"slug":101,"count":102,"latest_published_at":103},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]