[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-researchers-pinpoint-why-llms-know-answers-but-still-hallucinate":10,"sections":34},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":29,"feedback":33,"feedback_at":22,"cost_usd":33,"total_tokens":33},10073,"researchers-pinpoint-why-llms-know-answers-but-still-hallucinate","Researchers Pinpoint Why LLMs Know Answers but Still Hallucinate","New research shows AI models often encode the correct answer internally yet still output a different, wrong token at decision time.","New research shows large language models frequently know the right answer internally, then say something else entirely.\n\nResearchers probed several LLMs to separate two things that usually get lumped together: whether the correct answer is decodable from a model's internal, intermediate computations, and whether the model actually selects that answer as its output. Using three separate probing methods and a randomized-label control to rule out lucky guesses, they found that a substantial share of wrong answers were still decodable internally - the model had the right information, it just did not act on it. The gap traced back to what the paper calls the selection margin: the difference between how strongly the final output layer supports the correct answer versus its top competitor. Artificially restoring that support to levels seen in successful answers fixed the first token in most failure cases across most models tested, though the wrong competitor still won in many of the remaining cases.\n\nThat is a meaningful wrinkle in how hallucinations get talked about: a lot of wrong answers are not missing knowledge, they are a tug-of-war the output layer loses. The researchers also found that borrowing support from a successful phrasing of the same fact works mainly through the model's later internal layers, not through attention. The effect is limited, though - patching the first token rarely rescues the whole answer, which the paper treats as evidence that decodability, recoverability, and generation are three separate problems, not one.\n\nTranslation for anyone hoping for a quick hallucination fix: knowing a model has the right answer buried inside it is not the same as getting it to say so, and nudging its confidence rarely fixes more than the first word.","[\"ai\",\"hallucinations\",\"interpretability\",\"llms\"]","2026-10-05T04:00:00.000Z","2026-10-05T21:39:48.242Z","2026-10-05T21:39:53.505Z","published",null,[],"ai",[24,26,27,28],"hallucinations","interpretability","llms",[30],{"name":31,"url":32},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2603.13911",0,{"sections":35},[36,39,43,48,53,58,62,67,71,76,81,86,91,96],{"name":37,"slug":24,"count":38,"latest_published_at":18},"AI",6314,{"name":40,"slug":41,"count":42,"latest_published_at":18},"Security","security",871,{"name":44,"slug":45,"count":46,"latest_published_at":47},"Policy","policy",444,"2026-10-03T15:02:01.000Z",{"name":49,"slug":50,"count":51,"latest_published_at":52},"Deals","deals",325,"2026-10-04T13:00:00.000Z",{"name":54,"slug":55,"count":56,"latest_published_at":57},"Hardware","hardware",204,"2026-10-03T14:50:50.000Z",{"name":59,"slug":60,"count":61,"latest_published_at":18},"Science","science",179,{"name":63,"slug":64,"count":65,"latest_published_at":66},"Consumer Tech","consumer-tech",158,"2026-10-03T03:21:12.000Z",{"name":68,"slug":69,"count":70,"latest_published_at":18},"Dev Tools","dev-tools",98,{"name":72,"slug":73,"count":74,"latest_published_at":75},"Software","software",97,"2026-10-04T10:00:00.000Z",{"name":77,"slug":78,"count":79,"latest_published_at":80},"Startups","startups",92,"2026-10-04T14:36:25.000Z",{"name":82,"slug":83,"count":84,"latest_published_at":85},"Gaming","gaming",53,"2026-10-02T02:50:39.000Z",{"name":87,"slug":88,"count":89,"latest_published_at":90},"General","general",51,"2026-10-05T02:35:01.000Z",{"name":92,"slug":93,"count":94,"latest_published_at":95},"Reviews","reviews",32,"2026-10-02T18:00:00.000Z",{"name":97,"slug":98,"count":99,"latest_published_at":100},"How-To","how-to",7,"2026-10-01T09:00:00.000Z"]