[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-study-finds-ai-judges-wrongly-penalize-correct-agent-answers":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},8232,"study-finds-ai-judges-wrongly-penalize-correct-agent-answers","Study Finds AI Judges Wrongly Penalize Correct Agent Answers","A new benchmark finds that reference-based AI judges flag correct agent answers as errors when entity details differ, and offers only a partial fix.","A widely used way to grade AI agents turns out to flunk them for getting the right answer.\n\nResearchers built CARGO, a framework for evaluating AI agents that work on live, changing data - things like support tickets, accounts, or assets. The standard method, called LLM-as-a-judge, compares an agent's answer to a stored reference answer and treats any mismatch as an error - so when an agent correctly handles a different case than the one in the reference, the judge wrongly flags different names, dates, and IDs as mistakes, rejecting correct answers 100% of the time on their test set. CARGO instead treats the reference as a template, checks each claim against the live case's own facts, and only penalizes claims that actually contradict that context. On a 246-item benchmark, CARGO eliminated those false penalties (0 out of 50) while still catching nearly all genuine contradictions (50 of 50 and 49 of 50 across two judge models).\n\nThis matters because companies increasingly lean on LLM judges to decide which AI agents get deployed in support, finance, and operations work. If the judge can't tell a correctly handled new case from a genuine hallucination, teams either reject working agents or lose trust in the eval and ship without one. It's a reminder that most AI evaluation tooling was built for static quiz-style benchmarks, not the messy, constantly changing data real agents actually touch.\n\nCARGO isn't a clean win: it still misses 80% of cases where an agent follows the wrong procedure but lands on plausible-looking facts, and the same blind spot showed up when other AI systems were used as human-style annotators - proof that grading agents fairly is still mostly guesswork dressed up as a benchmark.","[\"ai evaluation\",\"llm-as-a-judge\",\"agentic ai\",\"benchmarks\"]","2026-09-28T04:00:00.000Z","2026-09-28T18:18:00.374Z","2026-09-28T18:18:06.877Z","published",null,[],"ai",[26,27,28,29],"ai evaluation","llm-as-a-judge","agentic ai","benchmarks",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2609.30471",0,{"sections":36},[37,40,44,49,54,59,63,68,73,78,83,88,92,97],{"name":38,"slug":24,"count":39,"latest_published_at":18},"AI",4844,{"name":41,"slug":42,"count":43,"latest_published_at":18},"Security","security",762,{"name":45,"slug":46,"count":47,"latest_published_at":48},"Policy","policy",399,"2026-09-27T18:39:02.000Z",{"name":50,"slug":51,"count":52,"latest_published_at":53},"Deals","deals",266,"2026-09-28T14:00:00.000Z",{"name":55,"slug":56,"count":57,"latest_published_at":58},"Hardware","hardware",189,"2026-09-28T10:52:40.000Z",{"name":60,"slug":61,"count":62,"latest_published_at":18},"Science","science",151,{"name":64,"slug":65,"count":66,"latest_published_at":67},"Consumer Tech","consumer-tech",135,"2026-09-26T14:30:00.000Z",{"name":69,"slug":70,"count":71,"latest_published_at":72},"Software","software",91,"2026-09-25T20:55:00.000Z",{"name":74,"slug":75,"count":76,"latest_published_at":77},"Dev Tools","dev-tools",84,"2026-09-26T04:20:58.000Z",{"name":79,"slug":80,"count":81,"latest_published_at":82},"Startups","startups",76,"2026-09-25T18:33:59.000Z",{"name":84,"slug":85,"count":86,"latest_published_at":87},"Gaming","gaming",48,"2026-09-25T18:35:21.000Z",{"name":89,"slug":90,"count":86,"latest_published_at":91},"General","general","2026-09-26T17:02:42.000Z",{"name":93,"slug":94,"count":95,"latest_published_at":96},"Reviews","reviews",30,"2026-09-24T20:07:31.000Z",{"name":98,"slug":99,"count":100,"latest_published_at":101},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]