[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-new-ai-framework-verifies-bugs-before-trying-to-fix-them":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},10109,"new-ai-framework-verifies-bugs-before-trying-to-fix-them","New AI Framework Verifies Bugs Before Trying to Fix Them","Researchers built a cross-language framework that forces AI to confirm a vulnerability is actually exploitable before it attempts a repair.","A new research framework makes AI agents prove a security bug is real before letting them patch it.\n\nThe system, described in a paper on arXiv, tackles vulnerability detection and repair across Java, Python, and C++ by converting all three into a shared structural format called a Universal Abstract Syntax Tree. It pairs a graph-based model (GraphSAGE) with code embeddings from Qwen2.5-Coder-1.5B, then runs detection, execution-based validation, and repair as three distinct stages. The hard rule: no automated fix happens until the tool actually confirms the flaw is exploitable, not just statistically likely. In testing, it hit 89.84-92.02% accuracy spotting bugs within a single language, 74.43-80.12% F1 on languages it wasn't trained on, and resolved 69.74% of confirmed vulnerabilities end to end, with a 12.27% overall failure rate.\n\nThis matters because AI coding agents are quietly being handed write access to real codebases, and most of them act on a classifier's best guess rather than checked evidence. The paper's own ablation makes the stakes concrete: turning off the validation step caused unnecessary repairs to jump 131.7%, which is the agentic-AI equivalent of a doctor operating on a hunch. That's the same failure mode showing up in AI-assisted coding generally - plausible-sounding output standing in for verified fact.\n\nStill, a 12% failure rate and an 80% ceiling on unfamiliar languages means this is a meaningful step, not a solved problem, and arXiv self-reported benchmarks deserve the usual grain of salt until independently replicated.","[\"ai agents\",\"vulnerability detection\",\"code security\",\"llm\"]","2026-10-05T04:00:00.000Z","2026-10-05T23:33:07.779Z","2026-10-05T23:33:14.171Z","published",null,[],"ai",[26,27,28,29],"ai agents","vulnerability detection","code security","llm",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2604.10800",0,{"sections":36},[37,41,45,50,55,60,64,69,74,79,84,89,94,99],{"name":38,"slug":24,"count":39,"latest_published_at":40},"AI",6317,"2026-10-05T09:51:57.000Z",{"name":42,"slug":43,"count":44,"latest_published_at":18},"Security","security",871,{"name":46,"slug":47,"count":48,"latest_published_at":49},"Policy","policy",446,"2026-10-05T10:25:00.000Z",{"name":51,"slug":52,"count":53,"latest_published_at":54},"Deals","deals",340,"2026-10-05T09:18:03.000Z",{"name":56,"slug":57,"count":58,"latest_published_at":59},"Hardware","hardware",205,"2026-10-05T10:58:22.000Z",{"name":61,"slug":62,"count":63,"latest_published_at":18},"Science","science",179,{"name":65,"slug":66,"count":67,"latest_published_at":68},"Consumer Tech","consumer-tech",160,"2026-10-05T10:23:15.000Z",{"name":70,"slug":71,"count":72,"latest_published_at":73},"Dev Tools","dev-tools",99,"2026-10-05T10:47:06.000Z",{"name":75,"slug":76,"count":77,"latest_published_at":78},"Software","software",97,"2026-10-04T10:00:00.000Z",{"name":80,"slug":81,"count":82,"latest_published_at":83},"Startups","startups",93,"2026-10-05T11:13:51.000Z",{"name":85,"slug":86,"count":87,"latest_published_at":88},"Gaming","gaming",53,"2026-10-02T02:50:39.000Z",{"name":90,"slug":91,"count":92,"latest_published_at":93},"General","general",51,"2026-10-05T02:35:01.000Z",{"name":95,"slug":96,"count":97,"latest_published_at":98},"Reviews","reviews",32,"2026-10-02T18:00:00.000Z",{"name":100,"slug":101,"count":102,"latest_published_at":103},"How-To","how-to",8,"2026-10-05T09:00:00.000Z"]