[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-new-math-predicts-when-ai-agents-can-fix-their-own-mistakes":10,"sections":33},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":28,"feedback":32,"feedback_at":22,"cost_usd":32,"total_tokens":32},10100,"new-math-predicts-when-ai-agents-can-fix-their-own-mistakes","New Math Predicts When AI Agents Can Fix Their Own Mistakes","Researchers show AI agents' ability to recover from failed tool calls follows a predictable formula, not luck or model size.","AI agents that bounce back from a failed tool call aren't improvising; they're following a measurable rule.\n\nA new paper introduces a metric called Expected Recovery Regret, or ERR, which measures how far an agent's recovery strategy strays from the optimal one when a tool call misfires. The researchers link ERR to a simpler, observable number they call the Efficiency Score, producing a first-order equation that predicts how much an agent's performance will suffer after a failure. They tested the law across five benchmarks, covering controlled error injection, diagnostic reasoning tasks, and live API calls. Predicted regret matched the regret actually observed in Monte Carlo simulations to within 0.05, across different model sizes and failure conditions.\n\nThat's the real finding here: recovery isn't a side effect of a bigger or smarter model. It's governed by the mechanics of the interaction itself, how the agent probes, retries, and adapts after something breaks. For anyone building agents that call external APIs or tools in production, that's a testable way to budget for failure instead of just hoping a bigger model shrugs it off.\n\nMost self-healing agent claims amount to anecdotes and cherry-picked demos; this one at least comes with an equation you could try to falsify.","[\"ai\",\"ai-agents\",\"research\"]","2026-10-05T04:00:00.000Z","2026-10-05T23:01:13.584Z","2026-10-05T23:01:19.314Z","published",null,[],"ai",[24,26,27],"ai-agents","research",[29],{"name":30,"url":31},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2601.22352",0,{"sections":34},[35,39,43,48,53,58,62,67,72,77,82,87,92,97],{"name":36,"slug":24,"count":37,"latest_published_at":38},"AI",6317,"2026-10-05T09:51:57.000Z",{"name":40,"slug":41,"count":42,"latest_published_at":18},"Security","security",871,{"name":44,"slug":45,"count":46,"latest_published_at":47},"Policy","policy",446,"2026-10-05T10:25:00.000Z",{"name":49,"slug":50,"count":51,"latest_published_at":52},"Deals","deals",340,"2026-10-05T09:18:03.000Z",{"name":54,"slug":55,"count":56,"latest_published_at":57},"Hardware","hardware",205,"2026-10-05T10:58:22.000Z",{"name":59,"slug":60,"count":61,"latest_published_at":18},"Science","science",179,{"name":63,"slug":64,"count":65,"latest_published_at":66},"Consumer Tech","consumer-tech",160,"2026-10-05T10:23:15.000Z",{"name":68,"slug":69,"count":70,"latest_published_at":71},"Dev Tools","dev-tools",99,"2026-10-05T10:47:06.000Z",{"name":73,"slug":74,"count":75,"latest_published_at":76},"Software","software",97,"2026-10-04T10:00:00.000Z",{"name":78,"slug":79,"count":80,"latest_published_at":81},"Startups","startups",93,"2026-10-05T11:13:51.000Z",{"name":83,"slug":84,"count":85,"latest_published_at":86},"Gaming","gaming",53,"2026-10-02T02:50:39.000Z",{"name":88,"slug":89,"count":90,"latest_published_at":91},"General","general",51,"2026-10-05T02:35:01.000Z",{"name":93,"slug":94,"count":95,"latest_published_at":96},"Reviews","reviews",32,"2026-10-02T18:00:00.000Z",{"name":98,"slug":99,"count":100,"latest_published_at":101},"How-To","how-to",8,"2026-10-05T09:00:00.000Z"]