[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-a-trivial-text-trick-jailbreaks-reasoning-ai-models":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},7737,"a-trivial-text-trick-jailbreaks-reasoning-ai-models","A Trivial Text Trick Jailbreaks Reasoning AI Models","A new study shows that feeding reasoning models a poisoned scratchpad plus a short forced reply opener can push jailbreak success as high as 99 percent.","Faking an AI's chain of thought does nothing by itself - but pair it with a one-line head start on the actual answer, and reasoning models cave up to 99 percent of the time.\n\nA new preprint tests the technique, called an output-prefix attack, on three 2026-era frontier models: Gemini 3 Flash Preview, DeepSeek V4 Flash, and Claude Haiku 4.5. The attack works because everything a model generates is conditioned on whatever text comes before it, including text an attacker slips into the start of the reply itself - something some APIs allow by exposing the model's intermediate reasoning channel for editing. Researchers ran a factorial test, three prefix types crossed with two reasoning-injection conditions, across 1,800 malicious prompts drawn from the AdvBench benchmark. Injecting poisoned fake reasoning alone barely worked, with close to a 0 percent success rate, but adding a trivial forced opening to the model's actual answer pushed success as high as 99 percent on some models, and prompts tailored to the specific request beat generic ones.\n\nThat's notable because reasoning models were sold as harder to trick, on the theory that an extra thinking step would catch bad requests before they got answered. This result suggests that scratchpad is beside the point if the final-answer channel can be pre-seeded - the model's chain of thought doesn't matter once its reply has already been forced open. It's the same prefix-injection weakness that has jailbroken ordinary chatbots for a while, just redirected at models built to look more careful.\n\nOne preprint isn't a verdict, but a 99 percent break rate on production-grade models is a hard number to shrug off - especially for any API that hands developers a raw reasoning channel to edit.","[\"ai-security\",\"jailbreak\",\"prompt-injection\",\"reasoning-models\"]","2026-09-25T04:00:00.000Z","2026-09-25T18:56:13.640Z","2026-09-25T18:56:19.907Z","published",null,[],"security",[26,27,28,29],"ai-security","jailbreak","prompt-injection","reasoning-models",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2609.29775",0,{"sections":36},[37,42,46,51,56,61,66,71,76,81,86,91,96,101],{"name":38,"slug":39,"count":40,"latest_published_at":41},"AI","ai",4482,"2026-09-25T15:40:03.000Z",{"name":43,"slug":24,"count":44,"latest_published_at":45},"Security",734,"2026-09-25T15:52:13.000Z",{"name":47,"slug":48,"count":49,"latest_published_at":50},"Policy","policy",388,"2026-09-25T15:27:35.000Z",{"name":52,"slug":53,"count":54,"latest_published_at":55},"Deals","deals",247,"2026-09-25T15:26:22.000Z",{"name":57,"slug":58,"count":59,"latest_published_at":60},"Hardware","hardware",183,"2026-09-25T13:41:27.000Z",{"name":62,"slug":63,"count":64,"latest_published_at":65},"Science","science",139,"2026-09-25T11:55:23.000Z",{"name":67,"slug":68,"count":69,"latest_published_at":70},"Consumer Tech","consumer-tech",132,"2026-09-25T15:30:00.000Z",{"name":72,"slug":73,"count":74,"latest_published_at":75},"Software","software",88,"2026-09-24T23:06:55.000Z",{"name":77,"slug":78,"count":79,"latest_published_at":80},"Dev Tools","dev-tools",81,"2026-09-25T09:59:40.000Z",{"name":82,"slug":83,"count":84,"latest_published_at":85},"Startups","startups",73,"2026-09-25T14:05:04.000Z",{"name":87,"slug":88,"count":89,"latest_published_at":90},"Gaming","gaming",47,"2026-09-25T13:33:16.000Z",{"name":92,"slug":93,"count":94,"latest_published_at":95},"General","general",46,"2026-09-25T02:12:57.000Z",{"name":97,"slug":98,"count":99,"latest_published_at":100},"Reviews","reviews",30,"2026-09-24T20:07:31.000Z",{"name":102,"slug":103,"count":104,"latest_published_at":105},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]