[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-study-finds-cheaper-ai-reviewers-can-audit-coding-agents":10,"sections":34},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":29,"feedback":33,"feedback_at":22,"cost_usd":33,"total_tokens":33},9697,"study-finds-cheaper-ai-reviewers-can-audit-coding-agents","Study Finds Cheaper AI Reviewers Can Audit Coding Agents","A new study finds that smaller AI reviewers can reliably judge whether coding agents' patches actually work, but only when given the right evidence to check.","A new study tests whether a cheaper AI reviewer can tell when a coding agent's patch only looks like it works.\n\nResearchers evaluated 411 execution-labeled traces from three coding agents, plus 101 controlled test cases. On 154 GPT-5.4 traces, giving a reviewer unverified structured evidence made it catch more real defects, but also wrongly reject more good patches. Given official execution evidence, the actual pass-fail results from running the code, as a best-case benchmark, five of six reviewer setups improved on both error rates across 122 held-out traces, and two judged every trace correctly. Reviewer model size did not reliably predict which ones performed better.\n\nThat ceiling matters because official test results are not available when these agents operate in the real world. So the researchers built a deployment-realistic fallback: checking for errors the patch introduces plus generated tests that failed before the patch was applied. On held-out GPT-5.4 and Gemini traces, that fallback caught 76 to 80 percent of bad patches, but still wrongly rejected good ones roughly two-thirds of the time, mostly on cases the agent never actually resolved.\n\nIn other words, grading whether AI-written code actually works, without rerunning a trusted test suite, is still mostly unsolved; this paper just measures the gap honestly.","[\"ai\",\"coding-agents\",\"llm-evaluation\",\"software-testing\"]","2026-10-02T04:00:00.000Z","2026-10-03T06:47:59.162Z","2026-10-03T06:48:04.105Z","published",null,[],"ai",[24,26,27,28],"coding-agents","llm-evaluation","software-testing",[30],{"name":31,"url":32},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2610.01023",0,{"sections":35},[36,39,43,47,52,56,60,65,70,75,80,85,90,95],{"name":37,"slug":24,"count":38,"latest_published_at":18},"AI",5989,{"name":40,"slug":41,"count":42,"latest_published_at":18},"Security","security",842,{"name":44,"slug":45,"count":46,"latest_published_at":18},"Policy","policy",438,{"name":48,"slug":49,"count":50,"latest_published_at":51},"Deals","deals",317,"2026-10-01T22:00:00.000Z",{"name":53,"slug":54,"count":55,"latest_published_at":18},"Hardware","hardware",199,{"name":57,"slug":58,"count":59,"latest_published_at":18},"Science","science",174,{"name":61,"slug":62,"count":63,"latest_published_at":64},"Consumer Tech","consumer-tech",155,"2026-10-01T19:54:10.000Z",{"name":66,"slug":67,"count":68,"latest_published_at":69},"Dev Tools","dev-tools",96,"2026-10-01T16:57:03.000Z",{"name":71,"slug":72,"count":73,"latest_published_at":74},"Software","software",93,"2026-09-30T21:41:11.000Z",{"name":76,"slug":77,"count":78,"latest_published_at":79},"Startups","startups",90,"2026-10-01T21:55:22.000Z",{"name":81,"slug":82,"count":83,"latest_published_at":84},"Gaming","gaming",53,"2026-10-02T02:50:39.000Z",{"name":86,"slug":87,"count":88,"latest_published_at":89},"General","general",50,"2026-09-30T21:37:54.000Z",{"name":91,"slug":92,"count":93,"latest_published_at":94},"Reviews","reviews",31,"2026-09-28T14:31:34.000Z",{"name":96,"slug":97,"count":98,"latest_published_at":99},"How-To","how-to",7,"2026-10-01T09:00:00.000Z"]