[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-new-framework-lets-ai-red-teaming-learn-from-its-failures":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},7543,"new-framework-lets-ai-red-teaming-learn-from-its-failures","New Framework Lets AI Red Teaming Learn From Its Failures","CART adapts its red-teaming prompts based on what already broke a model, finding more failures than static tests across text and agentic AI systems.","A new AI red teaming system stops replaying the same jailbreak scripts and starts learning from what worked.\n\nResearchers describe CART, short for Closed-Loop Adaptive Red Teaming, in a new paper. Instead of firing a fixed list of prompts at a model, CART uses each test result to decide what to try next, chasing weaknesses as they surface while keeping its probes varied enough to avoid tunnel vision. The system splits the job into three roles: a Challenger that writes the tests, a Target that gets attacked (either a plain text model or an AI agent restricted to a limited toolset), and a Judge that scores the results. Tested across three benchmark families, CART surfaced more failures and higher average risk scores than static prompt replay on every model with a baseline to compare against, including agents that use tools.\n\nThat matters because most automated red teaming still amounts to running the same known attack list on a schedule, which tells you whether a model still falls for last year's tricks and little else. CART's gains on tool-using agents are the more interesting result: it suggests adaptive probing can expose weaknesses that scripted prompts never reach, which matters as more products ship as agents rather than plain chatbots. The researchers also found that swapping which model plays Challenger or Judge changes what gets found, an argument for keeping those roles independently audited rather than run by the same system.\n\nThis paper measures what CART's test policies can find, not how often these failures show up in real deployment. A red team that gets smarter is not the same as a model that gets safer.","[\"ai-safety\",\"red-teaming\",\"llm-security\",\"ai\"]","2026-09-24T04:00:00.000Z","2026-09-24T05:23:32.265Z","2026-09-24T05:23:38.543Z","published",null,[],"security",[26,27,28,29],"ai-safety","red-teaming","llm-security","ai",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2609.27336",0,{"sections":36},[37,40,43,48,53,58,63,68,73,78,83,88,93,98],{"name":38,"slug":29,"count":39,"latest_published_at":18},"AI",4387,{"name":41,"slug":24,"count":42,"latest_published_at":18},"Security",720,{"name":44,"slug":45,"count":46,"latest_published_at":47},"Policy","policy",380,"2026-09-23T22:53:43.000Z",{"name":49,"slug":50,"count":51,"latest_published_at":52},"Deals","deals",220,"2026-09-23T22:00:00.000Z",{"name":54,"slug":55,"count":56,"latest_published_at":57},"Hardware","hardware",173,"2026-09-23T23:42:10.000Z",{"name":59,"slug":60,"count":61,"latest_published_at":62},"Science","science",135,"2026-09-23T22:45:49.000Z",{"name":64,"slug":65,"count":66,"latest_published_at":67},"Consumer Tech","consumer-tech",116,"2026-09-24T00:51:49.000Z",{"name":69,"slug":70,"count":71,"latest_published_at":72},"Software","software",85,"2026-09-23T20:00:00.000Z",{"name":74,"slug":75,"count":76,"latest_published_at":77},"Dev Tools","dev-tools",79,"2026-09-22T22:21:13.000Z",{"name":79,"slug":80,"count":81,"latest_published_at":82},"Startups","startups",66,"2026-09-23T17:28:38.000Z",{"name":84,"slug":85,"count":86,"latest_published_at":87},"Gaming","gaming",45,"2026-09-22T15:35:06.000Z",{"name":89,"slug":90,"count":91,"latest_published_at":92},"General","general",43,"2026-09-21T23:48:56.000Z",{"name":94,"slug":95,"count":96,"latest_published_at":97},"Reviews","reviews",27,"2026-09-22T13:00:00.000Z",{"name":99,"slug":100,"count":101,"latest_published_at":102},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]