[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-jailbreak-benchmarks-might-be-grading-confusion-not-safety":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},7839,"jailbreak-benchmarks-might-be-grading-confusion-not-safety","Jailbreak Benchmarks Might Be Grading Confusion, Not Safety","A new study finds refusal-rate benchmarks can't tell a genuinely safer model from one that has simply stopped understanding requests at all.","A widely used way of grading how well AI models resist jailbreaks may be measuring the wrong thing entirely.\n\nResearchers ran four 7-8B open models, spanning three base families and four post-training recipes, through a standard encoded-prompt jailbreak test, then did something most benchmarks skip: they ran the harmless version of the same test too. Refusal rates on harmful homoglyph-encoded prompts barely varied across models, but plaintext refusal rates spread widely, and on one model the gap between refusing bad requests and complying with good ones collapsed from +0.82 in plaintext to exactly 0.00 once the prompts were encoded. A harmful-arm-only benchmark scored that model identical to one that kept a +0.61 gap. Pushing a model through a full SFT to DPO to RLVR training pipeline actually raised the harm gap by +0.26, a real improvement the conventional metric registered as no change whatsoever.\n\nThis matters because harmful-arm-only scoring is the default way labs report safety gains after alignment training, and it can't distinguish a model that got better at judgment from one that just stopped parsing encoded text as a request at all. The researchers also catalogued twelve flaws in the measurement tools themselves, six of which make models look safer than they are, including a binary jailbreak judge that flagged 61 to 70 percent of benign plaintext replies as jailbreak attempts.\n\nA model that refuses everything isn't safe. It's just broken in a way that happens to look good on a scoreboard.","[\"ai safety\",\"jailbreak benchmarks\",\"llm evaluation\",\"encoded prompts\"]","2026-09-25T04:00:00.000Z","2026-09-26T01:39:01.532Z","2026-09-26T01:39:06.416Z","published",null,[],"ai",[26,27,28,29],"ai safety","jailbreak benchmarks","llm evaluation","encoded prompts",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2609.26176",0,{"sections":36},[37,41,46,51,56,61,66,71,76,81,86,91,96,101],{"name":38,"slug":24,"count":39,"latest_published_at":40},"AI",4557,"2026-09-25T17:16:30.000Z",{"name":42,"slug":43,"count":44,"latest_published_at":45},"Security","security",741,"2026-09-25T15:52:13.000Z",{"name":47,"slug":48,"count":49,"latest_published_at":50},"Policy","policy",390,"2026-09-25T16:24:59.000Z",{"name":52,"slug":53,"count":54,"latest_published_at":55},"Deals","deals",256,"2026-09-25T17:00:53.000Z",{"name":57,"slug":58,"count":59,"latest_published_at":60},"Hardware","hardware",185,"2026-09-25T15:00:22.000Z",{"name":62,"slug":63,"count":64,"latest_published_at":65},"Science","science",140,"2026-09-25T11:55:23.000Z",{"name":67,"slug":68,"count":69,"latest_published_at":70},"Consumer Tech","consumer-tech",132,"2026-09-25T15:30:00.000Z",{"name":72,"slug":73,"count":74,"latest_published_at":75},"Software","software",88,"2026-09-24T23:06:55.000Z",{"name":77,"slug":78,"count":79,"latest_published_at":80},"Dev Tools","dev-tools",82,"2026-09-25T09:59:40.000Z",{"name":82,"slug":83,"count":84,"latest_published_at":85},"Startups","startups",76,"2026-09-25T18:33:59.000Z",{"name":87,"slug":88,"count":89,"latest_published_at":90},"Gaming","gaming",48,"2026-09-25T18:35:21.000Z",{"name":92,"slug":93,"count":94,"latest_published_at":95},"General","general",46,"2026-09-25T02:12:57.000Z",{"name":97,"slug":98,"count":99,"latest_published_at":100},"Reviews","reviews",30,"2026-09-24T20:07:31.000Z",{"name":102,"slug":103,"count":104,"latest_published_at":105},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]