[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-new-system-automatically-patches-ai-agent-guardrails":10,"sections":33},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":28,"feedback":32,"feedback_at":22,"cost_usd":32,"total_tokens":32},9916,"new-system-automatically-patches-ai-agent-guardrails","New System Automatically Patches AI Agent Guardrails","HASTE uses dueling AI agents to rewrite an AI agent's safety rules from nothing more than a few attack examples, then keeps testing itself for new holes.","A new system patches AI agents' safety defenses on its own, using little more than scraps of threat intel.\n\nResearchers introduced HASTE, a multi-agent framework that updates agent harnesses - the guardrails that stop AI agents from taking unsafe actions - based on sparse evidence like brief attack descriptions or a handful of examples pulled from threat reports. HASTE pits two processes against each other: one generates safety specifications meant to close an identified gap, the other generates new attack cases designed to break that fix. The outcomes feed back into both sides, so the harness keeps tightening even against attacks it was never directly shown. Tested across multiple backbone models and attack types, the system cut attack success rates while keeping the agent able to do its normal job. Code is posted on GitHub.\n\nRight now, patching an agent's defenses after a new jailbreak surfaces is slow and manual: someone reads a report, edits a policy, redeploys. HASTE is a bet that this cycle can run automatically and still generalize beyond the exact attack it was shown, which matters because threat reports rarely contain enough detail to fully reconstruct an exploit.\n\nWhether an adversarial generator can reliably out-imagine real attackers, not just the ones it invents for itself, is the real test, and no benchmark can settle that until it meets attackers who never read the paper.","[\"ai-safety\",\"ai-agents\",\"security\"]","2026-10-05T04:00:00.000Z","2026-10-05T13:13:24.552Z","2026-10-05T13:13:30.882Z","published",null,[],"security",[26,27,24],"ai-safety","ai-agents",[29],{"name":30,"url":31},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2610.02920",0,{"sections":34},[35,39,42,47,52,57,61,66,70,74,79,84,89,94],{"name":36,"slug":37,"count":38,"latest_published_at":18},"AI","ai",6168,{"name":40,"slug":24,"count":41,"latest_published_at":18},"Security",859,{"name":43,"slug":44,"count":45,"latest_published_at":46},"Policy","policy",444,"2026-10-03T15:02:01.000Z",{"name":48,"slug":49,"count":50,"latest_published_at":51},"Deals","deals",323,"2026-10-04T13:00:00.000Z",{"name":53,"slug":54,"count":55,"latest_published_at":56},"Hardware","hardware",204,"2026-10-03T14:50:50.000Z",{"name":58,"slug":59,"count":60,"latest_published_at":18},"Science","science",177,{"name":62,"slug":63,"count":64,"latest_published_at":65},"Consumer Tech","consumer-tech",158,"2026-10-03T03:21:12.000Z",{"name":67,"slug":68,"count":69,"latest_published_at":18},"Dev Tools","dev-tools",97,{"name":71,"slug":72,"count":69,"latest_published_at":73},"Software","software","2026-10-04T10:00:00.000Z",{"name":75,"slug":76,"count":77,"latest_published_at":78},"Startups","startups",92,"2026-10-04T14:36:25.000Z",{"name":80,"slug":81,"count":82,"latest_published_at":83},"Gaming","gaming",53,"2026-10-02T02:50:39.000Z",{"name":85,"slug":86,"count":87,"latest_published_at":88},"General","general",51,"2026-10-05T02:35:01.000Z",{"name":90,"slug":91,"count":92,"latest_published_at":93},"Reviews","reviews",32,"2026-10-02T18:00:00.000Z",{"name":95,"slug":96,"count":97,"latest_published_at":98},"How-To","how-to",7,"2026-10-01T09:00:00.000Z"]