[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-angry-wording-alone-makes-ai-reasoning-models-slip-up":10,"sections":40},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":30,"tags":31,"sources":35,"feedback":39,"feedback_at":22,"cost_usd":39,"total_tokens":39},4926,"angry-wording-alone-makes-ai-reasoning-models-slip-up","Angry Wording Alone Makes AI Reasoning Models Slip Up","A new benchmark finds emotional phrasing alone, with no numbers changed, cuts AI reasoning accuracy by 2 to 10 percentage points.","Typing an angry version of your math homework into an AI chatbot can make it get the answer wrong, even if every number stays exactly the same.\n\nResearchers built a benchmark called Temper-5400 to isolate whether emotional wording alone degrades quantitative reasoning in large language models. They took problems from three standard benchmarks: the grade-school math sets GSM8K and MultiArith, plus the science-reasoning benchmark ARC-Challenge, and rewrote each into an emotional version, injecting frustration, urgency, or enthusiasm while keeping every quantity and relationship untouched. The resulting 5,400 verified pairs were tested on eighteen models spanning 1-billion-parameter systems up to frontier-scale ones. Accuracy fell by 2 to 10 percentage points on the emotional versions compared with the neutral originals.\n\nThat is a real swing for a change that alters zero facts, just tone. The team also found that stripping the emotional language back out at inference time recovered most of the lost accuracy, while plain non-emotional paraphrasing caused no degradation at all, which points squarely at emotional framing, not surface rewording, as the cause.\n\nReal users do not type in clean, emotionally neutral prose. A model that gets worse at arithmetic the moment you sound stressed is a usability problem hiding inside a tidy benchmark number.","[\"ai\",\"llm-evaluation\",\"benchmark\",\"prompt-engineering\"]","2026-08-13T04:00:00.000Z","2026-08-14T17:51:47.656Z","2026-08-14T17:51:58.695Z","published",null,[24],{"id":25,"reviewer":26,"round":27,"reason":28,"status":29},"publisher-r1","publisher",1,"ARC-Challenge is a science-reasoning QA benchmark, not a math problem set, so describing all three source benchmarks (GSM8K, MultiArith, ARC-Challenge) as 'standard math problems' is an internal factual inconsistency.","resolved","ai",[30,32,33,34],"llm-evaluation","benchmark","prompt-engineering",[36],{"name":37,"url":38},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2604.07801",0,{"sections":41},[42,46,50,55,60,65,70,75,80,85,90,95,100,105],{"name":43,"slug":30,"count":44,"latest_published_at":45},"AI",3293,"2026-08-20T04:00:00.000Z",{"name":47,"slug":48,"count":49,"latest_published_at":45},"Security","security",435,{"name":51,"slug":52,"count":53,"latest_published_at":54},"Policy","policy",210,"2026-08-19T09:32:27.000Z",{"name":56,"slug":57,"count":58,"latest_published_at":59},"Deals","deals",179,"2026-06-29T20:02:07.000Z",{"name":61,"slug":62,"count":63,"latest_published_at":64},"Hardware","hardware",140,"2026-08-19T18:25:42.000Z",{"name":66,"slug":67,"count":68,"latest_published_at":69},"Consumer Tech","consumer-tech",95,"2026-08-18T16:05:00.000Z",{"name":71,"slug":72,"count":73,"latest_published_at":74},"Science","science",90,"2026-08-19T18:41:02.000Z",{"name":76,"slug":77,"count":78,"latest_published_at":79},"Software","software",73,"2026-08-18T07:51:50.000Z",{"name":81,"slug":82,"count":83,"latest_published_at":84},"Dev Tools","dev-tools",69,"2026-08-18T04:00:00.000Z",{"name":86,"slug":87,"count":88,"latest_published_at":89},"Startups","startups",47,"2026-08-19T19:13:46.000Z",{"name":91,"slug":92,"count":93,"latest_published_at":94},"Gaming","gaming",41,"2026-07-09T04:00:00.000Z",{"name":96,"slug":97,"count":98,"latest_published_at":99},"General","general",33,"2026-08-18T22:18:13.000Z",{"name":101,"slug":102,"count":103,"latest_published_at":104},"Reviews","reviews",20,"2026-06-24T12:00:01.000Z",{"name":106,"slug":107,"count":108,"latest_published_at":109},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]