[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-new-method-measures-when-ai-agents-are-guessing":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},7300,"new-method-measures-when-ai-agents-are-guessing","New Method Measures When AI Agents Are Guessing","Researchers built a way to score how confident an AI agent's step-by-step reasoning actually is, aiming to flag shaky answers before they cause damage.","Researchers have a new way to tell when an AI agent is confidently making things up.\n\nThe method, called GRUET, targets so-called ReAct agents: systems that alternate between reasoning in text and taking actions, like searching the web or calling a tool, across multiple turns. Give the same task to one of these agents twice and you can get wildly different paths to an answer. The researchers argue that inconsistency traces back to uncertainty piling up turn by turn, so they built a graph of the possible reasoning branches at each step and used the graph's complexity as a proxy for how shaky that step really is. Add up those per-turn scores and you get a confidence rating for the whole trajectory. The team tested it across nine different LLMs and five benchmarks, checking whether it could reliably flag which answers to trust.\n\nThis matters because \"the agent seemed sure of itself\" is currently the closest thing most people have to a trust signal, and it's a bad one. Confident-sounding chains of thought and correct answers are not the same thing, and as agents get handed real tasks like booking travel or managing infrastructure, an unreliable trajectory that looks fine on the surface is the failure mode that actually costs money. A working uncertainty score means a system could flag or halt a risky run before it acts, rather than after.\n\nIt's not a fix for hallucination, just a smoke detector for it. Whether it holds up outside benchmark conditions, where agents chase messier goals with real consequences, is the open question.","[\"ai agents\",\"llm\",\"uncertainty quantification\",\"ai research\"]","2026-09-23T04:00:00.000Z","2026-09-23T06:03:25.145Z","2026-09-23T06:03:30.464Z","published",null,[],"ai",[26,27,28,29],"ai agents","llm","uncertainty quantification","ai research",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2609.24831",0,{"sections":36},[37,40,44,49,54,59,63,68,73,78,83,88,93,98],{"name":38,"slug":24,"count":39,"latest_published_at":18},"AI",4264,{"name":41,"slug":42,"count":43,"latest_published_at":18},"Security","security",707,{"name":45,"slug":46,"count":47,"latest_published_at":48},"Policy","policy",369,"2026-09-23T02:13:52.000Z",{"name":50,"slug":51,"count":52,"latest_published_at":53},"Deals","deals",202,"2026-09-22T23:00:04.000Z",{"name":55,"slug":56,"count":57,"latest_published_at":58},"Hardware","hardware",168,"2026-09-22T23:56:03.000Z",{"name":60,"slug":61,"count":62,"latest_published_at":18},"Science","science",133,{"name":64,"slug":65,"count":66,"latest_published_at":67},"Consumer Tech","consumer-tech",110,"2026-09-22T20:00:00.000Z",{"name":69,"slug":70,"count":71,"latest_published_at":72},"Software","software",80,"2026-09-22T23:32:52.000Z",{"name":74,"slug":75,"count":76,"latest_published_at":77},"Dev Tools","dev-tools",79,"2026-09-22T22:21:13.000Z",{"name":79,"slug":80,"count":81,"latest_published_at":82},"Startups","startups",65,"2026-09-22T22:06:48.000Z",{"name":84,"slug":85,"count":86,"latest_published_at":87},"Gaming","gaming",45,"2026-09-22T15:35:06.000Z",{"name":89,"slug":90,"count":91,"latest_published_at":92},"General","general",43,"2026-09-21T23:48:56.000Z",{"name":94,"slug":95,"count":96,"latest_published_at":97},"Reviews","reviews",27,"2026-09-22T13:00:00.000Z",{"name":99,"slug":100,"count":101,"latest_published_at":102},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]