[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-study-finds-written-ai-rubrics-miss-half-of-human-preference":10,"sections":34},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":29,"feedback":33,"feedback_at":22,"cost_usd":33,"total_tokens":33},9926,"study-finds-written-ai-rubrics-miss-half-of-human-preference","Study Finds Written AI Rubrics Miss Half of Human Preference","A new 2.8 million text dataset shows that even math and code preferences carry a tacit component no written rubric fully captures.","A massive new dataset suggests the rules we write for AI models to follow miss a lot of what people actually want.\n\nResearchers built CreativePreferences, a dataset of 2.8 million texts rated by 317 million human judgments across seven creative domains and 42 benchmark tasks. They trained three kinds of models on the same judgments: one built from explicit rule-based programs, one from rubric banks, and one trained directly on the dense human preference data. Comparing the three turned up consistent gaps - models trained on raw preference data beat the rubric-based ones at matching what humans actually chose, even in domains assumed to be objective, like mathematics and software engineering. The gaps were largest in peer review and creative writing, and grew wider as more human judges weighed in.\n\nThis matters because most current AI training leans on exactly the rubric-and-verifier approach the study pokes holes in - RLAIF and RLVR both assume that if a preference can be written down, it can be trained on directly. The researchers found the opposite dynamic at work too: once you articulate a preference as a rule, the model starts optimizing for the rule instead of the underlying judgment it was meant to approximate, a pattern they liken to Goodhart's law.\n\nThe uncomfortable detail is that math and code - the domains AI labs most often cite as proof that verifiable rewards work - showed the same blind spot as peer review and fiction.","[\"ai\",\"rlhf\",\"preference-learning\",\"research\"]","2026-10-05T04:00:00.000Z","2026-10-05T13:39:45.839Z","2026-10-05T13:39:52.355Z","published",null,[],"ai",[24,26,27,28],"rlhf","preference-learning","research",[30],{"name":31,"url":32},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2610.03025",0,{"sections":35},[36,39,43,48,53,58,62,67,71,75,80,85,90,95],{"name":37,"slug":24,"count":38,"latest_published_at":18},"AI",6167,{"name":40,"slug":41,"count":42,"latest_published_at":18},"Security","security",859,{"name":44,"slug":45,"count":46,"latest_published_at":47},"Policy","policy",444,"2026-10-03T15:02:01.000Z",{"name":49,"slug":50,"count":51,"latest_published_at":52},"Deals","deals",323,"2026-10-04T13:00:00.000Z",{"name":54,"slug":55,"count":56,"latest_published_at":57},"Hardware","hardware",204,"2026-10-03T14:50:50.000Z",{"name":59,"slug":60,"count":61,"latest_published_at":18},"Science","science",177,{"name":63,"slug":64,"count":65,"latest_published_at":66},"Consumer Tech","consumer-tech",158,"2026-10-03T03:21:12.000Z",{"name":68,"slug":69,"count":70,"latest_published_at":18},"Dev Tools","dev-tools",97,{"name":72,"slug":73,"count":70,"latest_published_at":74},"Software","software","2026-10-04T10:00:00.000Z",{"name":76,"slug":77,"count":78,"latest_published_at":79},"Startups","startups",92,"2026-10-04T14:36:25.000Z",{"name":81,"slug":82,"count":83,"latest_published_at":84},"Gaming","gaming",53,"2026-10-02T02:50:39.000Z",{"name":86,"slug":87,"count":88,"latest_published_at":89},"General","general",51,"2026-10-05T02:35:01.000Z",{"name":91,"slug":92,"count":93,"latest_published_at":94},"Reviews","reviews",32,"2026-10-02T18:00:00.000Z",{"name":96,"slug":97,"count":98,"latest_published_at":99},"How-To","how-to",7,"2026-10-01T09:00:00.000Z"]