[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-cutting-refusal-boilerplate-cuts-false-refusals-in-chatbots":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},6134,"cutting-refusal-boilerplate-cuts-false-refusals-in-chatbots","Cutting Refusal Boilerplate Cuts False Refusals in Chatbots","A new study finds language models trained without canned refusal phrases still refuse harmful requests but stop blocking harmless lookalikes.","A new study says the reason your chatbot refuses to help you \"shoot\" a photo has less to do with judgment and more to do with a fill-in-the-blank template baked into its training data.\n\nResearchers analyzed how safety-tuning datasets teach models to refuse harmful requests. Each training example typically pairs a boilerplate refusal line (\"I can't help with that\") with a rationale explaining why. The team found that training on both parts together teaches models to key off surface-level red-flag words rather than actual intent, which is why a query like \"where can I shoot a good photo\" can get flagged the same as \"how do I shoot someone.\" When they trained models on the rationale alone, stripping out the canned refusal phrasing, false refusals dropped while the models still caught genuinely harmful requests at a comparable rate. The rationale-only effect also held up in in-context learning setups and worked alongside existing inference-time safety techniques.\n\nThis matters because over-refusal is the quieter cousin of jailbreaking: it doesn't make headlines, but it's the daily friction that makes chatbots feel useless for anyone doing security research, medical writing, or just asking an oddly phrased question. Most safety work chases the dramatic failure mode, models that say yes to bad requests, while the boring failure mode, models that say no to fine ones, gets fixed with workarounds instead of better training data.\n\nIt's a reminder that a lot of AI safety behavior is less \"reasoning about harm\" and more pattern-matching on the training set's own habits. Fix the dataset's structure, and you fix a chunk of the behavior, no bigger model required.","[\"ai-safety\",\"language-models\",\"alignment\",\"llm-training\"]","2026-09-07T04:00:00.000Z","2026-09-07T06:47:12.285Z","2026-09-07T06:47:24.194Z","published",null,[],"ai",[26,27,28,29],"ai-safety","language-models","alignment","llm-training",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2609.04714",0,{"sections":36},[37,41,46,51,56,61,66,71,76,81,86,91,96,101],{"name":38,"slug":24,"count":39,"latest_published_at":40},"AI",3411,"2026-09-07T10:06:25.000Z",{"name":42,"slug":43,"count":44,"latest_published_at":45},"Security","security",574,"2026-09-07T07:03:20.000Z",{"name":47,"slug":48,"count":49,"latest_published_at":50},"Policy","policy",311,"2026-09-07T05:33:25.000Z",{"name":52,"slug":53,"count":54,"latest_published_at":55},"Deals","deals",179,"2026-06-29T20:02:07.000Z",{"name":57,"slug":58,"count":59,"latest_published_at":60},"Hardware","hardware",152,"2026-09-03T09:26:48.000Z",{"name":62,"slug":63,"count":64,"latest_published_at":65},"Consumer Tech","consumer-tech",97,"2026-09-04T15:29:18.000Z",{"name":67,"slug":68,"count":69,"latest_published_at":70},"Science","science",96,"2026-09-03T22:30:00.000Z",{"name":72,"slug":73,"count":74,"latest_published_at":75},"Software","software",73,"2026-08-18T07:51:50.000Z",{"name":77,"slug":78,"count":79,"latest_published_at":80},"Dev Tools","dev-tools",69,"2026-08-18T04:00:00.000Z",{"name":82,"slug":83,"count":84,"latest_published_at":85},"Startups","startups",54,"2026-09-04T23:36:14.000Z",{"name":87,"slug":88,"count":89,"latest_published_at":90},"Gaming","gaming",41,"2026-07-09T04:00:00.000Z",{"name":92,"slug":93,"count":94,"latest_published_at":95},"General","general",40,"2026-09-07T08:57:14.000Z",{"name":97,"slug":98,"count":99,"latest_published_at":100},"Reviews","reviews",20,"2026-06-24T12:00:01.000Z",{"name":102,"slug":103,"count":104,"latest_published_at":105},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]