[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-new-benchmark-shows-ai-evaluators-fake-good-reasoning":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},5812,"new-benchmark-shows-ai-evaluators-fake-good-reasoning","New Benchmark Shows AI Evaluators Fake Good Reasoning","A new benchmark shows AI evaluators can nail the right verdict while their underlying reasoning falls apart under scrutiny.","AI models are increasingly used to grade other AI models' work - and a new benchmark suggests those graders often can't explain themselves.\n\nResearchers formalize what they call \"judgment receipts\": the minimal set of evidence, rules, or authority a verdict actually rests on. They built ReasonBench, a policy and logical reasoning benchmark spanning 19,520 cases and 7,200 controls, to test whether evaluator models could reproduce those receipts when the underlying facts changed. In controlled, frozen tests, a small model called Qwen3-1.7B scored 98.41% on receipt accuracy and 96.99% on predicting how judgments should shift - numbers that looked close to solved. Reshuffle the same evidence, rules, and authority into meaning-preserving orders, though, and receipt recovery collapsed to 54.8% and 49.2%.\n\nThat gap matters because these evaluators are gatekeepers now: they approve agent actions, route items for human review, and generate feedback used to train other models. A model that gets the label right without a stable reason for it is a liability wearing a passing grade. Retraining on simple reordered cases patched surface consistency to 96.6% but made deeper reasoning prediction worse, not better.\n\nIn other words, teaching a model to survive one flavor of scrutiny doesn't teach it to reason - it teaches it to survive that flavor of scrutiny. Anyone grading AI systems on accuracy alone is measuring the wrong thing.","[\"ai-evaluation\",\"llm-reasoning\",\"benchmarks\",\"ai-safety\"]","2026-08-24T04:00:00.000Z","2026-08-24T05:01:20.285Z","2026-08-24T05:01:32.216Z","published",null,[],"ai",[26,27,28,29],"ai-evaluation","llm-reasoning","benchmarks","ai-safety",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2608.20938",0,{"sections":36},[37,41,45,50,55,60,65,70,75,80,85,90,95,100],{"name":38,"slug":24,"count":39,"latest_published_at":40},"AI",3325,"2026-08-24T09:09:31.000Z",{"name":42,"slug":43,"count":44,"latest_published_at":18},"Security","security",461,{"name":46,"slug":47,"count":48,"latest_published_at":49},"Policy","policy",218,"2026-08-23T19:30:00.000Z",{"name":51,"slug":52,"count":53,"latest_published_at":54},"Deals","deals",179,"2026-06-29T20:02:07.000Z",{"name":56,"slug":57,"count":58,"latest_published_at":59},"Hardware","hardware",145,"2026-08-22T21:25:33.000Z",{"name":61,"slug":62,"count":63,"latest_published_at":64},"Consumer Tech","consumer-tech",95,"2026-08-18T16:05:00.000Z",{"name":66,"slug":67,"count":68,"latest_published_at":69},"Science","science",91,"2026-08-20T10:01:48.000Z",{"name":71,"slug":72,"count":73,"latest_published_at":74},"Software","software",73,"2026-08-18T07:51:50.000Z",{"name":76,"slug":77,"count":78,"latest_published_at":79},"Dev Tools","dev-tools",69,"2026-08-18T04:00:00.000Z",{"name":81,"slug":82,"count":83,"latest_published_at":84},"Startups","startups",50,"2026-08-22T16:23:09.000Z",{"name":86,"slug":87,"count":88,"latest_published_at":89},"Gaming","gaming",41,"2026-07-09T04:00:00.000Z",{"name":91,"slug":92,"count":93,"latest_published_at":94},"General","general",33,"2026-08-18T22:18:13.000Z",{"name":96,"slug":97,"count":98,"latest_published_at":99},"Reviews","reviews",20,"2026-06-24T12:00:01.000Z",{"name":101,"slug":102,"count":103,"latest_published_at":104},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]