[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-new-method-filters-risky-prompts-before-they-reach-ai-models":10,"sections":41},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":30,"tags":31,"sources":36,"feedback":40,"feedback_at":22,"cost_usd":40,"total_tokens":40},5816,"new-method-filters-risky-prompts-before-they-reach-ai-models","New Method Filters Risky Prompts Before They Reach AI Models","Researchers built a lightweight relay that rewrites suspicious multimodal prompts before they hit a deployed model, no retraining or internal access required.","A new research framework lets AI labs patch multimodal chatbots against jailbreak attempts without ever retraining them.\n\nResearchers describe ReFrame, a training-free system that sits in front of a deployed multimodal large language model and intercepts risky prompts before they reach it. It runs two small, locally deployed AI agents: one gathers evidence about a prompt's risk and its legitimate use, the other rewrites the prompt into a safer version and decides whether an accompanying image should be passed through at all. The system never touches the target model's weights or internal workings, so it can be layered onto closed-source models that developers can't retrain or inspect directly. Across multiple models and benchmarks, the researchers report better jailbreak defense and safety awareness, plus fewer cases of the model refusing harmless requests.\n\nMost safety fixes require retraining a model or reading its internal states, options unavailable to anyone calling someone else's model through an API. ReFrame instead works like a filter bolted onto the outside, which matters as more products plug proprietary multimodal models into consumer-facing tools their builders don't fully control.\n\nAdd-on safety layers like this are easier to ship than to trust: they patch a fixed set of known tricks, not the reasoning gaps attackers keep finding new ways to exploit.","[\"ai-safety\",\"multimodal-ai\",\"jailbreak-defense\",\"llm-security\"]","2026-08-24T04:00:00.000Z","2026-08-24T05:18:40.672Z","2026-08-24T05:18:52.585Z","published",null,[24],{"id":25,"reviewer":26,"round":27,"reason":28,"status":29},"publisher-r1","publisher",1,"The body ends with an editorializing critique paragraph and no concluding\u002Fclosing line, so it reads as an unfinished fragment rather than a complete article.","resolved","ai",[32,33,34,35],"ai-safety","multimodal-ai","jailbreak-defense","llm-security",[37],{"name":38,"url":39},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2608.21100",0,{"sections":42},[43,47,51,56,61,66,71,76,81,86,91,96,101,106],{"name":44,"slug":30,"count":45,"latest_published_at":46},"AI",3325,"2026-08-24T09:09:31.000Z",{"name":48,"slug":49,"count":50,"latest_published_at":18},"Security","security",461,{"name":52,"slug":53,"count":54,"latest_published_at":55},"Policy","policy",218,"2026-08-23T19:30:00.000Z",{"name":57,"slug":58,"count":59,"latest_published_at":60},"Deals","deals",179,"2026-06-29T20:02:07.000Z",{"name":62,"slug":63,"count":64,"latest_published_at":65},"Hardware","hardware",145,"2026-08-22T21:25:33.000Z",{"name":67,"slug":68,"count":69,"latest_published_at":70},"Consumer Tech","consumer-tech",95,"2026-08-18T16:05:00.000Z",{"name":72,"slug":73,"count":74,"latest_published_at":75},"Science","science",91,"2026-08-20T10:01:48.000Z",{"name":77,"slug":78,"count":79,"latest_published_at":80},"Software","software",73,"2026-08-18T07:51:50.000Z",{"name":82,"slug":83,"count":84,"latest_published_at":85},"Dev Tools","dev-tools",69,"2026-08-18T04:00:00.000Z",{"name":87,"slug":88,"count":89,"latest_published_at":90},"Startups","startups",50,"2026-08-22T16:23:09.000Z",{"name":92,"slug":93,"count":94,"latest_published_at":95},"Gaming","gaming",41,"2026-07-09T04:00:00.000Z",{"name":97,"slug":98,"count":99,"latest_published_at":100},"General","general",33,"2026-08-18T22:18:13.000Z",{"name":102,"slug":103,"count":104,"latest_published_at":105},"Reviews","reviews",20,"2026-06-24T12:00:01.000Z",{"name":107,"slug":108,"count":109,"latest_published_at":110},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]