[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-new-rl-method-lets-a-robot-dog-switch-priorities-on-the-fly":10,"sections":34},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":29,"feedback":33,"feedback_at":22,"cost_usd":33,"total_tokens":33},9719,"new-rl-method-lets-a-robot-dog-switch-priorities-on-the-fly","New RL Method Lets a Robot Dog Switch Priorities on the Fly","A new reinforcement learning policy lets a Unitree Go2 quadruped trade off speed, stability, and energy efficiency at runtime, not training time.","Researchers have taught a robot dog to switch its priorities on command, without retraining its brain.\n\nA team describes PROMO, a reinforcement learning method that lets a single policy control a quadruped's locomotion based on a preference set at runtime rather than baked in during training. Normal RL controllers hardcode a fixed trade-off between things like tracking a command, staying stable, and conserving energy. PROMO instead treats that trade-off as a dial an operator can turn after deployment. In simulation, the team sampled 100 different preferences and found 67 produced genuinely distinct, non-dominated behaviors, with a 0.843 correlation between the stated preference and the resulting behavior. The policy also transferred zero-shot to a real Unitree Go2, where changing the preference alone cut energy use by up to 30.4%, tracking error by 38.7%, and peak body-tilt deviation by 59.0% compared to a balanced setting.\n\nThat matters because robot deployments rarely have one fixed job. A warehouse quadruped might need to prioritize battery life on a long patrol and precise tracking when threading a tight aisle. Today that usually means training or fine-tuning separate models for each scenario. PROMO suggests one policy could cover that whole range, with the trade-off set like a configuration option instead of a retraining job.\n\nThe catch is that this is one lab's simulation-to-one-robot result, not a deployed product, and 67 non-dominated behaviors out of 100 preferences measures diversity, not how well any single behavior holds up in messy real-world conditions. The code is open-source, so expect other labs to stress-test the claim before it becomes standard practice.","[\"robotics\",\"reinforcement-learning\",\"quadrupedal-robots\",\"ai\"]","2026-10-02T04:00:00.000Z","2026-10-03T07:43:47.634Z","2026-10-03T07:43:52.108Z","published",null,[],"ai",[26,27,28,24],"robotics","reinforcement-learning","quadrupedal-robots",[30],{"name":31,"url":32},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2610.01260",0,{"sections":35},[36,39,43,47,52,56,60,65,70,75,80,85,90,95],{"name":37,"slug":24,"count":38,"latest_published_at":18},"AI",6041,{"name":40,"slug":41,"count":42,"latest_published_at":18},"Security","security",848,{"name":44,"slug":45,"count":46,"latest_published_at":18},"Policy","policy",439,{"name":48,"slug":49,"count":50,"latest_published_at":51},"Deals","deals",317,"2026-10-01T22:00:00.000Z",{"name":53,"slug":54,"count":55,"latest_published_at":18},"Hardware","hardware",199,{"name":57,"slug":58,"count":59,"latest_published_at":18},"Science","science",176,{"name":61,"slug":62,"count":63,"latest_published_at":64},"Consumer Tech","consumer-tech",155,"2026-10-01T19:54:10.000Z",{"name":66,"slug":67,"count":68,"latest_published_at":69},"Dev Tools","dev-tools",96,"2026-10-01T16:57:03.000Z",{"name":71,"slug":72,"count":73,"latest_published_at":74},"Software","software",93,"2026-09-30T21:41:11.000Z",{"name":76,"slug":77,"count":78,"latest_published_at":79},"Startups","startups",90,"2026-10-01T21:55:22.000Z",{"name":81,"slug":82,"count":83,"latest_published_at":84},"Gaming","gaming",53,"2026-10-02T02:50:39.000Z",{"name":86,"slug":87,"count":88,"latest_published_at":89},"General","general",50,"2026-09-30T21:37:54.000Z",{"name":91,"slug":92,"count":93,"latest_published_at":94},"Reviews","reviews",31,"2026-09-28T14:31:34.000Z",{"name":96,"slug":97,"count":98,"latest_published_at":99},"How-To","how-to",7,"2026-10-01T09:00:00.000Z"]