[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-new-benchmark-tests-how-well-ai-agents-know-theyre-wrong":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},9324,"new-benchmark-tests-how-well-ai-agents-know-theyre-wrong","New Benchmark Tests How Well AI Agents Know They're Wrong","Argus, a new benchmark, finds that AI agent confidence scores rank risk well within a single model but barely transfer across different models or vendors.","A new benchmark says the confidence scores guiding AI agents that click around your screen are only trustworthy if you don't swap the model underneath them.\n\nResearchers built Argus, a benchmark that tests 27 uncertainty-quantification methods across four open-weight vision-language models and four GUI-grounding datasets, plus eight methods across three closed-source frontier vendors where internal signals like logits and attention maps aren't exposed. These \"computer-use agents\" translate a model's visual read of a screen into an actual click, so knowing when a click is likely wrong matters for both safety and for catching bad guesses before they fire. The methods tested ranged from logit-based scores and sampling-consistency checks to hidden-state density estimators and agents simply stating their own confidence in words.\n\nThe headline result: a method's ability to rank risky predictions against safe ones holds up well across datasets for the same model, with correlation scores as high as 0.969, but that ranking barely survives a jump to a different model family. Applied to closed-source vendors, the average correlation drops to just 0.08. Calibrating the disk-shaped safety margins drawn around a predicted click does shrink them by 40 to 60 percent, but only when the calibration data matches the real deployment setup; mismatches break the coverage guarantee.\n\nThat's a useful corrective for anyone assuming a single confidence metric travels well from one AI agent to the next vendor's model. Uncertainty estimates here look less like a universal safety feature and more like a setting you have to retune every time the underlying model changes.","[\"ai agents\",\"computer-use agents\",\"benchmarks\",\"model evaluation\"]","2026-10-01T04:00:00.000Z","2026-10-02T09:27:11.515Z","2026-10-02T09:27:16.684Z","published",null,[],"ai",[26,27,28,29],"ai agents","computer-use agents","benchmarks","model evaluation",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2606.25760",0,{"sections":36},[37,41,46,51,56,61,65,70,75,79,83,88,93,98],{"name":38,"slug":24,"count":39,"latest_published_at":40},"AI",5694,"2026-10-01T12:05:27.000Z",{"name":42,"slug":43,"count":44,"latest_published_at":45},"Security","security",821,"2026-10-01T14:00:00.000Z",{"name":47,"slug":48,"count":49,"latest_published_at":50},"Policy","policy",431,"2026-10-01T11:08:42.000Z",{"name":52,"slug":53,"count":54,"latest_published_at":55},"Deals","deals",311,"2026-10-01T14:19:00.000Z",{"name":57,"slug":58,"count":59,"latest_published_at":60},"Hardware","hardware",197,"2026-10-01T11:37:06.000Z",{"name":62,"slug":63,"count":64,"latest_published_at":18},"Science","science",165,{"name":66,"slug":67,"count":68,"latest_published_at":69},"Consumer Tech","consumer-tech",150,"2026-10-01T11:59:27.000Z",{"name":71,"slug":72,"count":73,"latest_published_at":74},"Dev Tools","dev-tools",93,"2026-10-01T02:30:48.000Z",{"name":76,"slug":77,"count":73,"latest_published_at":78},"Software","software","2026-09-30T21:41:11.000Z",{"name":80,"slug":81,"count":82,"latest_published_at":45},"Startups","startups",86,{"name":84,"slug":85,"count":86,"latest_published_at":87},"Gaming","gaming",51,"2026-09-30T16:24:30.000Z",{"name":89,"slug":90,"count":91,"latest_published_at":92},"General","general",50,"2026-09-30T21:37:54.000Z",{"name":94,"slug":95,"count":96,"latest_published_at":97},"Reviews","reviews",31,"2026-09-28T14:31:34.000Z",{"name":99,"slug":100,"count":101,"latest_published_at":102},"How-To","how-to",7,"2026-10-01T09:00:00.000Z"]