[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-new-rl-method-keeps-ai-policies-from-collapsing-during-fine-tuning":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},11134,"new-rl-method-keeps-ai-policies-from-collapsing-during-fine-tuning","New RL Method Keeps AI Policies From Collapsing During Fine-Tuning","A new off-policy RL method prevents critic errors from collapsing flow-model fine-tuning, lifting success rates from 46% to 68% across 50 benchmark tasks.","A new reinforcement learning technique keeps AI models from falling apart when they're fine-tuned with a reward signal.\n\nThe method, called Trust Region Q Adjoint Matching (TRQAM), builds on a prior approach named Q-learning with Adjoint Matching (QAM), which treats the fine-tuning of pretrained flow policies as a stochastic optimal control problem guided by a learned critic. The problem is that critics make mistakes, and QAM has no way to stop those mistakes from snowballing into total performance collapse. TRQAM fixes this by capping how far the fine-tuned policy is allowed to drift from the original pretrained one, using a parameter that directly controls that distance in the underlying math. Tested on 50 tasks from the OGBench benchmark suite, TRQAM hit a 68% overall success rate in offline RL, compared with 46% for the best prior method.\n\nThat 22-point jump matters because critic-guided fine-tuning is the backbone of how labs try to make generative models better at specific tasks without retraining them from scratch, from robotics control to image generation. A critic that quietly feeds bad gradients into a policy is a failure mode that's easy to miss until a model's outputs degrade, and bolting on a hard limit on how far a policy can wander is a pragmatic fix rather than a flashy one.\n\nStill, this is a benchmark result, not a production deployment. Whether a fixed KL budget holds up on messier, real-world reward signals is the next question worth asking.","[\"reinforcement-learning\",\"flow-models\",\"ai-research\",\"benchmarks\"]","2026-10-09T04:00:00.000Z","2026-10-10T07:43:37.609Z","2026-10-10T07:43:42.821Z","published",null,[],"ai",[26,27,28,29],"reinforcement-learning","flow-models","ai-research","benchmarks",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2605.27079",0,{"sections":36},[37,41,46,51,56,61,66,71,76,81,86,91,96,101],{"name":38,"slug":24,"count":39,"latest_published_at":40},"AI",6840,"2026-10-09T19:36:56.000Z",{"name":42,"slug":43,"count":44,"latest_published_at":45},"Security","security",941,"2026-10-09T19:25:43.000Z",{"name":47,"slug":48,"count":49,"latest_published_at":50},"Policy","policy",490,"2026-10-09T14:25:43.000Z",{"name":52,"slug":53,"count":54,"latest_published_at":55},"Deals","deals",483,"2026-10-09T11:20:39.000Z",{"name":57,"slug":58,"count":59,"latest_published_at":60},"Hardware","hardware",233,"2026-10-09T16:07:54.000Z",{"name":62,"slug":63,"count":64,"latest_published_at":65},"Science","science",195,"2026-10-09T11:00:57.000Z",{"name":67,"slug":68,"count":69,"latest_published_at":70},"Consumer Tech","consumer-tech",183,"2026-10-09T14:50:57.000Z",{"name":72,"slug":73,"count":74,"latest_published_at":75},"Startups","startups",120,"2026-10-09T17:02:31.000Z",{"name":77,"slug":78,"count":79,"latest_published_at":80},"Software","software",114,"2026-10-08T17:57:01.000Z",{"name":82,"slug":83,"count":84,"latest_published_at":85},"Dev Tools","dev-tools",109,"2026-10-09T13:03:48.000Z",{"name":87,"slug":88,"count":89,"latest_published_at":90},"General","general",69,"2026-10-09T19:10:07.000Z",{"name":92,"slug":93,"count":94,"latest_published_at":95},"Gaming","gaming",59,"2026-10-09T11:43:43.000Z",{"name":97,"slug":98,"count":99,"latest_published_at":100},"Reviews","reviews",34,"2026-10-08T14:00:22.000Z",{"name":102,"slug":103,"count":104,"latest_published_at":105},"How-To","how-to",8,"2026-10-05T09:00:00.000Z"]