[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-new-ai-framework-cuts-missed-security-alerts-from-40-to-3":10,"sections":34},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":29,"feedback":33,"feedback_at":22,"cost_usd":33,"total_tokens":33},10940,"new-ai-framework-cuts-missed-security-alerts-from-40-to-3","New AI Framework Cuts Missed Security Alerts From 40% to 3%","Researchers found leading AI triage agents missed over 40 percent of attack alerts, but a new adversarial review system cut that to 3.1 percent.","AI agents built to triage security alerts miss more than four in ten real attacks - until you make them argue with themselves.\n\nResearchers tested five common designs for LLM-based alert triage agents, from single-pass tool use to iterative retrieval and self-review, against a benchmark called ALERT-BENCH that replays real enterprise telemetry through a live SIEM. Across 1,247 alerts tied to a multi-stage attack scenario, every single approach missed at least 40.4% of the alerts that actually mattered. The failure pattern was consistent: agents dismissed alerts whenever a search turned up no matching records, having the model review its own work produced no real correction, and alerts marked for dismissal got no more scrutiny than ones escalated to a human. The researchers then built AIDA, a multi-agent setup that forces a proposed verdict to survive an independent challenge in a separate reasoning context, logs every step in an append-only Investigation Ledger, and sends the dispute to a separate judge model that can demand another round of evidence before closing anything.\n\nThat jump - from a 40.4% false-negative rate down to 3.1%, while escalating 18.4% of alerts to human analysts - is the difference between an AI tool that quietly rubber-stamps breaches and one that's actually useful in a SOC. It's also a pointed data point in the broader debate over agentic AI: a model reasoning in one pass is not the same as a model forced to defend its reasoning against a skeptic.\n\nEscalating nearly a fifth of alerts isn't free - someone still has to work that queue - but it beats the alternative, which is an AI that looks decisive right up until the breach it waved through shows up in an incident report.","[\"ai-agents\",\"security\",\"soc\",\"llm\"]","2026-10-09T04:00:00.000Z","2026-10-09T22:38:16.791Z","2026-10-09T22:38:18.817Z","published",null,[],"security",[26,24,27,28],"ai-agents","soc","llm",[30],{"name":31,"url":32},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2610.10608",0,{"sections":35},[36,40,43,48,53,57,61,66,71,76,81,86,91,96],{"name":37,"slug":38,"count":39,"latest_published_at":18},"AI","ai",6708,{"name":41,"slug":24,"count":42,"latest_published_at":18},"Security",931,{"name":44,"slug":45,"count":46,"latest_published_at":47},"Policy","policy",486,"2026-10-08T22:40:11.000Z",{"name":49,"slug":50,"count":51,"latest_published_at":52},"Deals","deals",474,"2026-10-08T22:00:00.000Z",{"name":54,"slug":55,"count":56,"latest_published_at":18},"Hardware","hardware",231,{"name":58,"slug":59,"count":60,"latest_published_at":18},"Science","science",192,{"name":62,"slug":63,"count":64,"latest_published_at":65},"Consumer Tech","consumer-tech",181,"2026-10-08T23:26:35.000Z",{"name":67,"slug":68,"count":69,"latest_published_at":70},"Startups","startups",117,"2026-10-08T16:45:00.000Z",{"name":72,"slug":73,"count":74,"latest_published_at":75},"Software","software",114,"2026-10-08T17:57:01.000Z",{"name":77,"slug":78,"count":79,"latest_published_at":80},"Dev Tools","dev-tools",105,"2026-10-07T16:59:11.000Z",{"name":82,"slug":83,"count":84,"latest_published_at":85},"General","general",66,"2026-10-09T04:46:11.000Z",{"name":87,"slug":88,"count":89,"latest_published_at":90},"Gaming","gaming",58,"2026-10-08T20:08:45.000Z",{"name":92,"slug":93,"count":94,"latest_published_at":95},"Reviews","reviews",34,"2026-10-08T14:00:22.000Z",{"name":97,"slug":98,"count":99,"latest_published_at":100},"How-To","how-to",8,"2026-10-05T09:00:00.000Z"]