[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-rl-fine-tuning-makes-coding-agents-finish-the-last-mile":10,"sections":34},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":29,"feedback":33,"feedback_at":22,"cost_usd":33,"total_tokens":33},9680,"rl-fine-tuning-makes-coding-agents-finish-the-last-mile","RL Fine-Tuning Makes Coding Agents Finish the Last Mile","A reinforcement-learning tweak to an open-weight coding model cut corner-cutting failures and boosted scores on six benchmarks it never trained on.","A reinforcement-learning tweak taught a coding agent to stop declaring victory too early - and the lesson stuck even on tests it never saw.\n\nThe model is Kimi K2.7 Code, an open-weight, 1-trillion-parameter mixture-of-experts system with 32 billion active parameters. Researchers ran reinforcement learning using an algorithm called GSPO, applied through a lightweight adapter (LoRA) rather than retraining the full model, on 1,700 expert-built tasks: 1,000 repository-fix jobs checked by hidden pass\u002Ffail tests plus tests confirming nothing else broke, and 700 terminal tasks graded by expert-written verifiers. The reward dropped to zero whenever the agent's fix broke previously-working behavior. After one training pass, pass rates rose on all six external benchmarks tested, including jumps from 67.4 to 82.0 on Terminal-Bench 2.1 and from 5.0 to 25.0 on SWE-Marathon.\n\nThe interesting part is where the gains held up. Three of the six benchmarks were published after the training data was collected, and the model still improved on them with statistical significance, as it did under two agent harnesses it had never encountered during training. That points to something more useful than memorized patterns: the training appears to have taught a general habit of checking your own work - not skipping the parts of a task that don't show up in an obvious test - which is exactly the kind of last-mile sloppiness that makes agentic coding tools unreliable in practice.\n\nWorth keeping in perspective, though: Terminal-Bench 3 and 4 scores still landed at 12.1 percent and 7.6 percent after training, so a big relative jump from near-zero is progress, not proof the last mile is solved.","[\"ai\",\"reinforcement-learning\",\"coding-agents\",\"benchmarks\"]","2026-10-02T04:00:00.000Z","2026-10-03T06:01:11.128Z","2026-10-03T06:01:15.236Z","published",null,[],"ai",[24,26,27,28],"reinforcement-learning","coding-agents","benchmarks",[30],{"name":31,"url":32},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2610.00890",0,{"sections":35},[36,39,43,47,52,56,60,65,70,75,80,85,90,95],{"name":37,"slug":24,"count":38,"latest_published_at":18},"AI",5976,{"name":40,"slug":41,"count":42,"latest_published_at":18},"Security","security",842,{"name":44,"slug":45,"count":46,"latest_published_at":18},"Policy","policy",438,{"name":48,"slug":49,"count":50,"latest_published_at":51},"Deals","deals",317,"2026-10-01T22:00:00.000Z",{"name":53,"slug":54,"count":55,"latest_published_at":18},"Hardware","hardware",199,{"name":57,"slug":58,"count":59,"latest_published_at":18},"Science","science",173,{"name":61,"slug":62,"count":63,"latest_published_at":64},"Consumer Tech","consumer-tech",155,"2026-10-01T19:54:10.000Z",{"name":66,"slug":67,"count":68,"latest_published_at":69},"Dev Tools","dev-tools",96,"2026-10-01T16:57:03.000Z",{"name":71,"slug":72,"count":73,"latest_published_at":74},"Software","software",93,"2026-09-30T21:41:11.000Z",{"name":76,"slug":77,"count":78,"latest_published_at":79},"Startups","startups",90,"2026-10-01T21:55:22.000Z",{"name":81,"slug":82,"count":83,"latest_published_at":84},"Gaming","gaming",53,"2026-10-02T02:50:39.000Z",{"name":86,"slug":87,"count":88,"latest_published_at":89},"General","general",50,"2026-09-30T21:37:54.000Z",{"name":91,"slug":92,"count":93,"latest_published_at":94},"Reviews","reviews",31,"2026-09-28T14:31:34.000Z",{"name":96,"slug":97,"count":98,"latest_published_at":99},"How-To","how-to",7,"2026-10-01T09:00:00.000Z"]