[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-new-loss-function-aims-to-curb-robot-policy-instability":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},10986,"new-loss-function-aims-to-curb-robot-policy-instability","New Loss Function Aims to Curb Robot Policy Instability","A new loss term adds higher order action supervision to robot policies, aiming to curb the instability plaguing imitation and reinforcement learning.","Robots and self driving cars keep twitching and failing in ways that look fine in simulation but fall apart in the real world. A new paper argues it knows why.\n\nResearchers behind a paper posted to arXiv say imitation learning and reinforcement learning policies become unstable because they are trained to match only the immediate action label, what the paper calls a zeroth-order action, while ignoring how that action changes moment to moment. Their fix is a loss scheme that also supervises this first-order action information, adding temporal consistency without changing a policy's underlying architecture. The method is designed as a plug-and-play module that bolts onto deterministic, stochastic, or flow-based policies and slots into existing offline RL frameworks. On the OGBench and D4RL benchmarks, the authors report better performance, steadier control, and improved out-of-distribution generalization, particularly when training data is scarce.\n\nControl instability is the unglamorous reason a lot of promising RL and imitation learning research never makes it into a real robot or vehicle. A loss term you can drop into an existing model, rather than a new architecture you have to adopt wholesale, is a more practical fix if it actually works. The low-data generalization gains matter too, since that is precisely where deployed robot learning systems tend to break outside the lab.\n\nPlenty of papers promise to fix RL's fragility. Whether this one holds up outside curated benchmarks like D4RL and OGBench is the harder test.","[\"reinforcement-learning\",\"imitation-learning\",\"robotics\",\"offline-rl\"]","2026-10-09T04:00:00.000Z","2026-10-10T00:44:21.638Z","2026-10-10T00:44:27.670Z","published",null,[],"ai",[26,27,28,29],"reinforcement-learning","imitation-learning","robotics","offline-rl",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2610.11175",0,{"sections":36},[37,40,44,49,54,58,62,67,72,77,82,87,92,97],{"name":38,"slug":24,"count":39,"latest_published_at":18},"AI",6708,{"name":41,"slug":42,"count":43,"latest_published_at":18},"Security","security",931,{"name":45,"slug":46,"count":47,"latest_published_at":48},"Policy","policy",486,"2026-10-08T22:40:11.000Z",{"name":50,"slug":51,"count":52,"latest_published_at":53},"Deals","deals",474,"2026-10-08T22:00:00.000Z",{"name":55,"slug":56,"count":57,"latest_published_at":18},"Hardware","hardware",231,{"name":59,"slug":60,"count":61,"latest_published_at":18},"Science","science",192,{"name":63,"slug":64,"count":65,"latest_published_at":66},"Consumer Tech","consumer-tech",181,"2026-10-08T23:26:35.000Z",{"name":68,"slug":69,"count":70,"latest_published_at":71},"Startups","startups",117,"2026-10-08T16:45:00.000Z",{"name":73,"slug":74,"count":75,"latest_published_at":76},"Software","software",114,"2026-10-08T17:57:01.000Z",{"name":78,"slug":79,"count":80,"latest_published_at":81},"Dev Tools","dev-tools",105,"2026-10-07T16:59:11.000Z",{"name":83,"slug":84,"count":85,"latest_published_at":86},"General","general",66,"2026-10-09T04:46:11.000Z",{"name":88,"slug":89,"count":90,"latest_published_at":91},"Gaming","gaming",58,"2026-10-08T20:08:45.000Z",{"name":93,"slug":94,"count":95,"latest_published_at":96},"Reviews","reviews",34,"2026-10-08T14:00:22.000Z",{"name":98,"slug":99,"count":100,"latest_published_at":101},"How-To","how-to",8,"2026-10-05T09:00:00.000Z"]