[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-shrinking-ai-reasoning-models-doesnt-make-them-safer":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},7349,"shrinking-ai-reasoning-models-doesnt-make-them-safer","Shrinking AI Reasoning Models Doesn't Make Them Safer","A new study finds efficiency tricks like quantization and pruning cut jailbreak success mainly by breaking reasoning, not by improving alignment.","Shrinking a reasoning model to cut costs can look like a safety win. It isn't.\n\nResearchers ran the first broad study of how efficiency tricks like quantization and pruning affect large reasoning models' resistance to jailbreak attempts. Smaller, cheaper models do block more jailbreak attempts than their full-size counterparts, but the improvement is superficial. The team traced the drop to broken reasoning, not better judgment: the models lose the ability to follow a malicious request through to a coherent answer, producing what the researchers call \"attempted but failed\" responses. A look at the models' internal representations confirmed it: as reasoning ability degrades, models can't hold onto a malicious line of thought long enough to finish it.\n\nThat matters for anyone treating a lower jailbreak success rate as proof a compressed model is safer. The paper draws a hard line between real alignment and what it calls capability-induced failure, and argues only the former should count as safety. Of the methods tested, combining quantization with pruning came closest to cutting costs without just making the model too confused to misbehave.\n\nA model too confused to finish a bad idea isn't aligned. It's just broken in a way that happens to be convenient for a safety benchmark.","[\"ai-safety\",\"large-reasoning-models\",\"quantization\",\"jailbreaks\"]","2026-09-23T04:00:00.000Z","2026-09-23T08:52:46.331Z","2026-09-23T08:52:50.283Z","published",null,[],"ai",[26,27,28,29],"ai-safety","large-reasoning-models","quantization","jailbreaks",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2609.23587",0,{"sections":36},[37,40,44,49,54,58,62,67,72,77,82,87,92,97],{"name":38,"slug":24,"count":39,"latest_published_at":18},"AI",4297,{"name":41,"slug":42,"count":43,"latest_published_at":18},"Security","security",710,{"name":45,"slug":46,"count":47,"latest_published_at":48},"Policy","policy",369,"2026-09-23T02:13:52.000Z",{"name":50,"slug":51,"count":52,"latest_published_at":53},"Deals","deals",202,"2026-09-22T23:00:04.000Z",{"name":55,"slug":56,"count":57,"latest_published_at":18},"Hardware","hardware",169,{"name":59,"slug":60,"count":61,"latest_published_at":18},"Science","science",133,{"name":63,"slug":64,"count":65,"latest_published_at":66},"Consumer Tech","consumer-tech",110,"2026-09-22T20:00:00.000Z",{"name":68,"slug":69,"count":70,"latest_published_at":71},"Software","software",80,"2026-09-22T23:32:52.000Z",{"name":73,"slug":74,"count":75,"latest_published_at":76},"Dev Tools","dev-tools",79,"2026-09-22T22:21:13.000Z",{"name":78,"slug":79,"count":80,"latest_published_at":81},"Startups","startups",65,"2026-09-22T22:06:48.000Z",{"name":83,"slug":84,"count":85,"latest_published_at":86},"Gaming","gaming",45,"2026-09-22T15:35:06.000Z",{"name":88,"slug":89,"count":90,"latest_published_at":91},"General","general",43,"2026-09-21T23:48:56.000Z",{"name":93,"slug":94,"count":95,"latest_published_at":96},"Reviews","reviews",27,"2026-09-22T13:00:00.000Z",{"name":98,"slug":99,"count":100,"latest_published_at":101},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]