[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-new-training-method-ranks-ai-answers-instead-of-scoring-them":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},7345,"new-training-method-ranks-ai-answers-instead-of-scoring-them","New Training Method Ranks AI Answers Instead of Scoring Them","A new reinforcement learning technique swaps error-prone rubric score averaging for ranking, improving results across 16 benchmarks and three model sizes.","Researchers have proposed a fix for a subtle flaw in how AI labs train models to follow multi-part quality rubrics.\n\nReinforcement Learning with Verifiable Rewards, or RLVR, works well for tasks like math and code, where an answer is simply right or wrong. Labs are trying to extend it to fuzzier tasks judged against multi-part rubrics, but that means squashing several separate quality scores into the single number a training algorithm needs. The standard approach - normalizing each score and averaging them - assumes a strong result on one criterion can cancel out a weak one on another, and that scores across different rubrics are directly comparable. A new paper, RLVR2, argues that assumption falls apart when criteria are genuinely different in kind, and proposes ranking a model's answers within each criterion instead of averaging raw scores, then combining those rankings into one training signal.\n\nThat distinction matters because reward design quietly decides what a model learns to prioritize. Averaging lets a model ace an easy criterion to paper over failure on a harder one - the same reward-hacking problem that dogged early RLHF systems, just relocated to multi-criterion grading. The paper reports RLVR2 beating standard rubric-based training across three model sizes and 16 benchmarks, and says it curbs models' tendency to pad responses or format their way around the rubric instead of actually improving.\n\nIt's a plumbing fix, not a new capability. But plumbing like this is what decides whether the next wave of \"more helpful\" model releases reflects real progress or just a model that has gotten better at gaming its own report card.","[\"reinforcement learning\",\"ai research\",\"llm training\",\"benchmarks\"]","2026-09-23T04:00:00.000Z","2026-09-23T08:41:35.474Z","2026-09-23T08:41:41.688Z","published",null,[],"ai",[26,27,28,29],"reinforcement learning","ai research","llm training","benchmarks",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2609.23457",0,{"sections":36},[37,40,44,49,54,58,62,67,72,77,82,87,92,97],{"name":38,"slug":24,"count":39,"latest_published_at":18},"AI",4297,{"name":41,"slug":42,"count":43,"latest_published_at":18},"Security","security",710,{"name":45,"slug":46,"count":47,"latest_published_at":48},"Policy","policy",369,"2026-09-23T02:13:52.000Z",{"name":50,"slug":51,"count":52,"latest_published_at":53},"Deals","deals",202,"2026-09-22T23:00:04.000Z",{"name":55,"slug":56,"count":57,"latest_published_at":18},"Hardware","hardware",169,{"name":59,"slug":60,"count":61,"latest_published_at":18},"Science","science",133,{"name":63,"slug":64,"count":65,"latest_published_at":66},"Consumer Tech","consumer-tech",110,"2026-09-22T20:00:00.000Z",{"name":68,"slug":69,"count":70,"latest_published_at":71},"Software","software",80,"2026-09-22T23:32:52.000Z",{"name":73,"slug":74,"count":75,"latest_published_at":76},"Dev Tools","dev-tools",79,"2026-09-22T22:21:13.000Z",{"name":78,"slug":79,"count":80,"latest_published_at":81},"Startups","startups",65,"2026-09-22T22:06:48.000Z",{"name":83,"slug":84,"count":85,"latest_published_at":86},"Gaming","gaming",45,"2026-09-22T15:35:06.000Z",{"name":88,"slug":89,"count":90,"latest_published_at":91},"General","general",43,"2026-09-21T23:48:56.000Z",{"name":93,"slug":94,"count":95,"latest_published_at":96},"Reviews","reviews",27,"2026-09-22T13:00:00.000Z",{"name":98,"slug":99,"count":100,"latest_published_at":101},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]