[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-benchmark-shows-wide-gaps-in-how-ai-models-recall-facts":10,"sections":39},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":30,"tags":31,"sources":34,"feedback":38,"feedback_at":22,"cost_usd":38,"total_tokens":38},7882,"benchmark-shows-wide-gaps-in-how-ai-models-recall-facts","Benchmark Shows Wide Gaps in How AI Models Recall Facts","A new test of 18 language models finds accuracy on basic facts varies wildly, and simple wording changes can flip answers entirely.","A new benchmark called PROOF pokes holes in how confidently language models recite facts.\n\nResearchers built PROOF by turning a frozen snapshot of Wikidata into 18,486 multiple-choice questions covering 11,779 facts across 101 classes, 392 properties, and 14 domains. Every question includes an \"I don't know\" option, and roughly 1,849 are deliberately unanswerable \"no correct option\" traps. The team ran 18 open-weight models through 166,374 prompts each, then reworded a subset of questions nine different ways and altered decoding settings to see if answers held up. They also tested what happens when a false answer gets planted directly in the question.\n\nBase accuracy ranged from 6.58% to 57.59%. Only the weakest model actually landed below the 8.64% you'd expect from random guessing - the strongest model cleared that floor by a wide margin, so this isn't a story about models that don't know anything. It's a story about how unevenly they know it: every single model showed a 19 to 36 percentage-point swing in accuracy across domains, and neutral rewordings alone could shift scores by up to 26.5 points. Adversarial phrasing broke as much as 79.4% of previously correct answers, and confidence scores often stayed high even when the answer was wrong.\n\nA single \"factuality\" number on a model card was never telling you the whole story. PROOF just makes the gaps easier to see, and harder to wave away as noise.","[\"ai\",\"benchmarks\",\"language models\"]","2026-09-25T04:00:00.000Z","2026-09-26T04:28:18.389Z","2026-09-26T04:28:24.431Z","published",null,[24],{"id":25,"reviewer":26,"round":27,"reason":28,"status":29},"editor-r1","editor",1,"Fix the line claiming accuracy ranged down to 6.58%, 'well below the 8.64% you'd get from random guessing' — the range's top end, 57.59%, is far above chance, so as written the sentence misrepresents the data; only the worst-performing model falls below chance, and that should be stated separately and clearly.","resolved","ai",[30,32,33],"benchmarks","language models",[35],{"name":36,"url":37},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2609.29504",0,{"sections":40},[41,45,50,55,60,65,70,75,80,85,90,95,100,105],{"name":42,"slug":30,"count":43,"latest_published_at":44},"AI",4589,"2026-09-25T21:57:05.000Z",{"name":46,"slug":47,"count":48,"latest_published_at":49},"Security","security",744,"2026-09-25T21:09:27.000Z",{"name":51,"slug":52,"count":53,"latest_published_at":54},"Policy","policy",392,"2026-09-25T18:44:30.000Z",{"name":56,"slug":57,"count":58,"latest_published_at":59},"Deals","deals",256,"2026-09-25T17:00:53.000Z",{"name":61,"slug":62,"count":63,"latest_published_at":64},"Hardware","hardware",185,"2026-09-25T15:00:22.000Z",{"name":66,"slug":67,"count":68,"latest_published_at":69},"Science","science",142,"2026-09-25T14:07:46.000Z",{"name":71,"slug":72,"count":73,"latest_published_at":74},"Consumer Tech","consumer-tech",132,"2026-09-25T15:30:00.000Z",{"name":76,"slug":77,"count":78,"latest_published_at":79},"Software","software",90,"2026-09-25T20:55:00.000Z",{"name":81,"slug":82,"count":83,"latest_published_at":84},"Dev Tools","dev-tools",82,"2026-09-25T09:59:40.000Z",{"name":86,"slug":87,"count":88,"latest_published_at":89},"Startups","startups",76,"2026-09-25T18:33:59.000Z",{"name":91,"slug":92,"count":93,"latest_published_at":94},"Gaming","gaming",48,"2026-09-25T18:35:21.000Z",{"name":96,"slug":97,"count":98,"latest_published_at":99},"General","general",46,"2026-09-25T02:12:57.000Z",{"name":101,"slug":102,"count":103,"latest_published_at":104},"Reviews","reviews",30,"2026-09-24T20:07:31.000Z",{"name":106,"slug":107,"count":108,"latest_published_at":109},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]