[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-new-rts-ai-separates-strategy-picking-from-unit-control":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},10888,"new-rts-ai-separates-strategy-picking-from-unit-control","New RTS AI Separates Strategy Picking From Unit Control","A MicroRTS research system uses a bandit algorithm to switch strategies mid-game, beating top opponents more often than a single fixed-strategy policy.","A new AI system for real-time strategy games separates picking a game plan from executing it, and it wins more often because of that split.\n\nResearchers built a reinforcement learning agent for MicroRTS, a stripped-down strategy game used as an AI testbed. The system splits into two layers: an executor trained with Proximal Policy Optimization that follows discrete commands covering economy, army composition, military posture and worker behavior, and a strategist that uses a Thompson-sampling bandit to pick which commands to issue based on what it observes the opponent doing. That strategist never identifies who it's playing against, it just watches and adapts. Tested against a single flat PPO policy trained with the same budget, architecture, curriculum and self-play league, the layered system won significantly more often against three of the four strongest opponents, including both held-out ones never seen in training, with win rates climbing as high as 0.97.\n\nThe point isn't the game itself. Deep RL agents are notoriously brittle outside their training distribution, a weakness that has dogged game-playing bots long after their headline-grabbing debuts. Decoupling what to do from how to do it is a plausible fix: it lets one well-trained execution policy get reused across strategies, rather than retraining a whole new brain every time an opponent breaks the pattern.\n\nMicroRTS is a toy compared to StarCraft II or Dota 2, and a bandit swapping between a handful of preset command tuples is a long way from genuine strategic reasoning. Whether this approach holds up at real-game complexity is the actual question, and this paper doesn't answer it.","[\"reinforcement-learning\",\"microrts\",\"game-ai\",\"strategy-games\"]","2026-10-09T04:00:00.000Z","2026-10-09T20:09:20.763Z","2026-10-09T20:09:24.595Z","published",null,[],"ai",[26,27,28,29],"reinforcement-learning","microrts","game-ai","strategy-games",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2610.11663",0,{"sections":36},[37,40,44,49,54,59,63,68,73,78,83,88,93,98],{"name":38,"slug":24,"count":39,"latest_published_at":18},"AI",6619,{"name":41,"slug":42,"count":43,"latest_published_at":18},"Security","security",926,{"name":45,"slug":46,"count":47,"latest_published_at":48},"Policy","policy",486,"2026-10-08T22:40:11.000Z",{"name":50,"slug":51,"count":52,"latest_published_at":53},"Deals","deals",474,"2026-10-08T22:00:00.000Z",{"name":55,"slug":56,"count":57,"latest_published_at":58},"Hardware","hardware",229,"2026-10-08T20:47:10.000Z",{"name":60,"slug":61,"count":62,"latest_published_at":18},"Science","science",192,{"name":64,"slug":65,"count":66,"latest_published_at":67},"Consumer Tech","consumer-tech",181,"2026-10-08T23:26:35.000Z",{"name":69,"slug":70,"count":71,"latest_published_at":72},"Startups","startups",117,"2026-10-08T16:45:00.000Z",{"name":74,"slug":75,"count":76,"latest_published_at":77},"Software","software",114,"2026-10-08T17:57:01.000Z",{"name":79,"slug":80,"count":81,"latest_published_at":82},"Dev Tools","dev-tools",105,"2026-10-07T16:59:11.000Z",{"name":84,"slug":85,"count":86,"latest_published_at":87},"General","general",66,"2026-10-09T04:46:11.000Z",{"name":89,"slug":90,"count":91,"latest_published_at":92},"Gaming","gaming",58,"2026-10-08T20:08:45.000Z",{"name":94,"slug":95,"count":96,"latest_published_at":97},"Reviews","reviews",34,"2026-10-08T14:00:22.000Z",{"name":99,"slug":100,"count":101,"latest_published_at":102},"How-To","how-to",8,"2026-10-05T09:00:00.000Z"]