[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-asking-an-ai-to-explain-itself-makes-it-worse-at-predicting":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},6301,"asking-an-ai-to-explain-itself-makes-it-worse-at-predicting","Asking an AI to Explain Itself Makes It Worse at Predicting","A new study finds language models rank customer outcomes more accurately when scored directly than when asked to explain first.","Asking a customer-behavior model to explain its answer before giving one makes that answer less trustworthy, new research shows.\n\nA new study fine-tuned language models to predict customer behavior, like churn or purchase decisions, across four retail tasks in three markets, covering 13 model-domain combinations, two of them using fully public data and checkpoints. Researchers compared two ways of reading out a model's confidence: scoring the probability of an answer token directly, versus prompting the model to write a rationale first and then generate a verdict. The scored version ranked outcomes more accurately in 12 of the 13 cells, with gains of 1.5 to 14.5 points of AUC and a sign-test p-value around 0.003. How big the gap got depended on training: it turned slightly negative for an untuned base model and widened to nearly 14 points for a model specifically trained to produce rationales.\n\nThat matters because plenty of production systems ask models to think out loud before answering, on the assumption that the explanation is basically free evidence of the same underlying judgment. Examining roughly 9,000 of the generated rationales, the researchers found the model leans less on its single strongest predictive signal and drifts toward generic stock phrasing once it starts narrating, which drags down the answer that follows. A third method, asking for a bare probability before any verdict, fixed calibration dramatically (Brier score from 0.47 to 0.15) but only matched scoring's ranking accuracy for outcomes the model had seen enough of during training.\n\nThe paper's own recommendation splits the difference: keep the generated rationale for the human reading it, but pull the actual risk score from the scoring head, not from the model's narrated conclusion.","[\"llm evaluation\",\"model calibration\",\"customer analytics\",\"ai research\"]","2026-09-11T04:00:00.000Z","2026-09-11T05:13:49.130Z","2026-09-11T05:14:01.032Z","published",null,[],"ai",[26,27,28,29],"llm evaluation","model calibration","customer analytics","ai research",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2609.09882",0,{"sections":36},[37,40,44,48,53,58,63,66,71,75,80,85,90,95],{"name":38,"slug":24,"count":39,"latest_published_at":18},"AI",3507,{"name":41,"slug":42,"count":43,"latest_published_at":18},"Security","security",636,{"name":45,"slug":46,"count":47,"latest_published_at":18},"Policy","policy",338,{"name":49,"slug":50,"count":51,"latest_published_at":52},"Deals","deals",179,"2026-06-29T20:02:07.000Z",{"name":54,"slug":55,"count":56,"latest_published_at":57},"Hardware","hardware",153,"2026-09-09T15:12:32.000Z",{"name":59,"slug":60,"count":61,"latest_published_at":62},"Consumer Tech","consumer-tech",99,"2026-09-09T17:27:33.000Z",{"name":64,"slug":65,"count":61,"latest_published_at":18},"Science","science",{"name":67,"slug":68,"count":69,"latest_published_at":70},"Software","software",75,"2026-09-10T20:41:21.000Z",{"name":72,"slug":73,"count":74,"latest_published_at":18},"Dev Tools","dev-tools",70,{"name":76,"slug":77,"count":78,"latest_published_at":79},"Startups","startups",55,"2026-09-09T23:14:29.000Z",{"name":81,"slug":82,"count":83,"latest_published_at":84},"Gaming","gaming",43,"2026-09-10T12:18:06.000Z",{"name":86,"slug":87,"count":88,"latest_published_at":89},"General","general",41,"2026-09-08T01:57:23.000Z",{"name":91,"slug":92,"count":93,"latest_published_at":94},"Reviews","reviews",20,"2026-06-24T12:00:01.000Z",{"name":96,"slug":97,"count":98,"latest_published_at":99},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]