[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-new-reward-model-learns-multiple-valid-answers-instead-of-one":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},8730,"new-reward-model-learns-multiple-valid-answers-instead-of-one","New Reward Model Learns Multiple Valid Answers Instead of One","Researchers built a diffusion-based reward model that captures the many valid judgments behind an AI response instead of collapsing them into one number.","A new reward model for AI training skips the single score and hands back a whole spread of plausible judgments instead.\n\nResearchers built DRM, a Diffusion Reward Model, to fix a basic flaw in how AI systems learn from human feedback. Most reward models take a prompt and response and boil human judgment down to one number or one preset statistical shape, even though people often disagree reasonably about the same answer. DRM instead uses a frozen language model encoder paired with a small diffusion transformer - the same kind of denoising process behind image generators - to turn random noise into a reward vector, with no assumption about what shape that distribution should take. The same architecture handles both attribute scoring and head-to-head preference comparisons, and at inference time it draws many samples to build an empirical distribution that can be read as an average, a spread, or specific confidence bounds.\n\nThat distinction matters because reward models are the referee in reinforcement learning from human feedback, and a referee that only ever gives one verdict cannot tell you when a call was close. Across five benchmarks, DRM matched or beat baselines of the same size and training data, stayed competitive with much bigger reward models, and recovered multi-modal reward patterns that conventional single-output models flattened into a single point. When the researchers used DRM's uncertainty estimates to reject low-confidence calls or apply a lower-confidence-bound penalty, the resulting policies trained better than those guided by a standard scalar reward.\n\nIt's a small-scale academic result, not a production reward model deployed at frontier-lab scale, but the underlying complaint - that flattening disagreement into one score is a modeling choice, not a law of nature - is worth remembering next time a chatbot's alignment gets credited to a single clean number.","[\"reward-models\",\"rlhf\",\"diffusion-models\",\"ai-alignment\"]","2026-09-30T04:00:00.000Z","2026-09-30T22:21:02.359Z","2026-09-30T22:21:07.549Z","published",null,[],"ai",[26,27,28,29],"reward-models","rlhf","diffusion-models","ai-alignment",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2609.33803",0,{"sections":36},[37,41,45,49,54,59,63,68,73,77,82,87,92,97],{"name":38,"slug":24,"count":39,"latest_published_at":40},"AI",5567,"2026-10-01T04:00:00.000Z",{"name":42,"slug":43,"count":44,"latest_published_at":40},"Security","security",815,{"name":46,"slug":47,"count":48,"latest_published_at":40},"Policy","policy",430,{"name":50,"slug":51,"count":52,"latest_published_at":53},"Deals","deals",298,"2026-09-30T21:00:26.000Z",{"name":55,"slug":56,"count":57,"latest_published_at":58},"Hardware","hardware",196,"2026-09-30T13:00:00.000Z",{"name":60,"slug":61,"count":62,"latest_published_at":40},"Science","science",163,{"name":64,"slug":65,"count":66,"latest_published_at":67},"Consumer Tech","consumer-tech",149,"2026-09-30T22:57:11.000Z",{"name":69,"slug":70,"count":71,"latest_published_at":72},"Dev Tools","dev-tools",93,"2026-10-01T02:30:48.000Z",{"name":74,"slug":75,"count":71,"latest_published_at":76},"Software","software","2026-09-30T21:41:11.000Z",{"name":78,"slug":79,"count":80,"latest_published_at":81},"Startups","startups",84,"2026-09-30T20:39:09.000Z",{"name":83,"slug":84,"count":85,"latest_published_at":86},"Gaming","gaming",51,"2026-09-30T16:24:30.000Z",{"name":88,"slug":89,"count":90,"latest_published_at":91},"General","general",50,"2026-09-30T21:37:54.000Z",{"name":93,"slug":94,"count":95,"latest_published_at":96},"Reviews","reviews",31,"2026-09-28T14:31:34.000Z",{"name":98,"slug":99,"count":100,"latest_published_at":101},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]