[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-a-fix-for-the-instability-plaguing-llm-post-training":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":24,"persona_id":22,"persona_name":22,"section":25,"tags":26,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},7070,"a-fix-for-the-instability-plaguing-llm-post-training","A Fix for the Instability Plaguing LLM Post-Training","GVPO offers a mathematically guaranteed alternative to unstable importance-sampling methods in reinforcement learning based LLM post-training.","A new paper claims to fix one of reinforcement learning's most persistent headaches: the training instability that importance sampling introduces into LLM post-training.\n\nThe method, called Group Variance Policy Optimization (GVPO), builds on Group Relative Policy Optimization (GRPO), a post-training technique that compares a model's outputs against a group average. GRPO's reliance on importance sampling, a statistical reweighting trick, can make training runs unstable or hard to reproduce. GVPO replaces that trick with a formula derived directly from the math of reward maximization, which the authors say guarantees a single, well-defined optimal solution instead of a noisy approximation. They also show the same formula extends to on-policy distillation, where a smaller model learns by mimicking a larger one's live outputs.\n\nThat combination matters because post-training is where labs now spend significant effort refining reasoning ability, and unstable training runs waste compute and researcher time. A method that pairs a stability guarantee with a unified approach to distillation could simplify pipelines that currently rely on separate techniques for each job.\n\nIt's one arXiv preprint with no independent benchmarks yet, so treat the stability guarantee as promising math, not proven practice, until someone outside the author list reproduces it.","[\"ai\",\"post-training\",\"reinforcement-learning\",\"llms\"]","2026-09-21T04:00:00.000Z","2026-09-21T04:52:00.496Z","2026-09-21T04:52:11.977Z","published",null,[],"https:\u002F\u002Fcdn.xyz.onl\u002Farticle-images\u002Fa-fix-for-the-instability-plaguing-llm-post-training.webp","ai",[25,27,28,29],"post-training","reinforcement-learning","llms",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2609.21432",0,{"sections":36},[37,40,44,49,54,59,64,69,74,79,84,89,94,99],{"name":38,"slug":25,"count":39,"latest_published_at":18},"AI",4157,{"name":41,"slug":42,"count":43,"latest_published_at":18},"Security","security",679,{"name":45,"slug":46,"count":47,"latest_published_at":48},"Policy","policy",350,"2026-09-20T20:32:43.000Z",{"name":50,"slug":51,"count":52,"latest_published_at":53},"Deals","deals",179,"2026-06-29T20:02:07.000Z",{"name":55,"slug":56,"count":57,"latest_published_at":58},"Hardware","hardware",156,"2026-09-19T11:00:00.000Z",{"name":60,"slug":61,"count":62,"latest_published_at":63},"Science","science",130,"2026-09-20T13:48:11.000Z",{"name":65,"slug":66,"count":67,"latest_published_at":68},"Consumer Tech","consumer-tech",99,"2026-09-09T17:27:33.000Z",{"name":70,"slug":71,"count":72,"latest_published_at":73},"Dev Tools","dev-tools",78,"2026-09-18T04:00:00.000Z",{"name":75,"slug":76,"count":77,"latest_published_at":78},"Software","software",75,"2026-09-10T20:41:21.000Z",{"name":80,"slug":81,"count":82,"latest_published_at":83},"Startups","startups",55,"2026-09-09T23:14:29.000Z",{"name":85,"slug":86,"count":87,"latest_published_at":88},"Gaming","gaming",43,"2026-09-10T12:18:06.000Z",{"name":90,"slug":91,"count":92,"latest_published_at":93},"General","general",42,"2026-09-18T22:35:10.000Z",{"name":95,"slug":96,"count":97,"latest_published_at":98},"Reviews","reviews",20,"2026-06-24T12:00:01.000Z",{"name":100,"slug":101,"count":102,"latest_published_at":103},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]