[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-a-fix-for-rl-training-when-every-answer-attempt-fails":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},8104,"a-fix-for-rl-training-when-every-answer-attempt-fails","A Fix for RL Training When Every Answer Attempt Fails","New research called SORT teaches reasoning models from failed attempts by comparing token probabilities with and without a reference plan.","A new technique teaches AI reasoning models something even when every one of their practice attempts is dead wrong.\n\nReinforcement learning with verifiable rewards helps sharpen reasoning models, but the popular GRPO-style approach stalls on hard prompts: when every sampled rollout gets the answer wrong, there is no reward signal to learn from. A new method called Selective Off-Policy Reference Tuning (SORT) patches that gap with a repair update for exactly those all-wrong cases, without changing how rollouts are generated in the first place. It derives a plan from the reference solution, then compares token probabilities with and without that plan present. Tokens that become noticeably more predictable once the plan is added get weighted more heavily during training, turning a failed batch into a structured, partial learning signal instead of forcing the model to copy the reference answer word for word.\n\nThat distinction, selective and structure-aware rather than pure imitation, is the actual contribution. Copying reference solutions token for token tends to make models mimic surface patterns without understanding, which is especially useless on prompts they clearly don't grasp yet. Across three model backbones and eight reasoning benchmarks, SORT beat both plain GRPO and other guidance baselines, with the largest gains showing up on weaker models, precisely the ones that generate the most all-wrong batches to begin with.\n\nThe paper landed as a v3 replace on arXiv, a reminder that even heavily cited RL recipes like GRPO still carry obvious gaps that get quietly patched well after the initial hype cycle.","[\"reinforcement-learning\",\"reasoning-models\",\"llm-training\",\"grpo\"]","2026-09-28T04:00:00.000Z","2026-09-28T09:38:32.974Z","2026-09-28T09:38:39.859Z","published",null,[],"ai",[26,27,28,29],"reinforcement-learning","reasoning-models","llm-training","grpo",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2605.11505",0,{"sections":36},[37,40,44,49,54,59,63,68,73,78,83,88,92,97],{"name":38,"slug":24,"count":39,"latest_published_at":18},"AI",4791,{"name":41,"slug":42,"count":43,"latest_published_at":18},"Security","security",762,{"name":45,"slug":46,"count":47,"latest_published_at":48},"Policy","policy",399,"2026-09-27T18:39:02.000Z",{"name":50,"slug":51,"count":52,"latest_published_at":53},"Deals","deals",261,"2026-09-27T15:30:35.000Z",{"name":55,"slug":56,"count":57,"latest_published_at":58},"Hardware","hardware",188,"2026-09-27T20:46:36.000Z",{"name":60,"slug":61,"count":62,"latest_published_at":18},"Science","science",151,{"name":64,"slug":65,"count":66,"latest_published_at":67},"Consumer Tech","consumer-tech",135,"2026-09-26T14:30:00.000Z",{"name":69,"slug":70,"count":71,"latest_published_at":72},"Software","software",91,"2026-09-25T20:55:00.000Z",{"name":74,"slug":75,"count":76,"latest_published_at":77},"Dev Tools","dev-tools",84,"2026-09-26T04:20:58.000Z",{"name":79,"slug":80,"count":81,"latest_published_at":82},"Startups","startups",76,"2026-09-25T18:33:59.000Z",{"name":84,"slug":85,"count":86,"latest_published_at":87},"Gaming","gaming",48,"2026-09-25T18:35:21.000Z",{"name":89,"slug":90,"count":86,"latest_published_at":91},"General","general","2026-09-26T17:02:42.000Z",{"name":93,"slug":94,"count":95,"latest_published_at":96},"Reviews","reviews",30,"2026-09-24T20:07:31.000Z",{"name":98,"slug":99,"count":100,"latest_published_at":101},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]