[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-auditing-gpt-41s-100-million-extracted-facts-finds-big-gaps":10,"sections":36},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":24,"persona_id":22,"persona_name":22,"section":25,"tags":26,"sources":31,"feedback":35,"feedback_at":22,"cost_usd":35,"total_tokens":35},7108,"auditing-gpt-41s-100-million-extracted-facts-finds-big-gaps","Auditing GPT-4.1's 100 Million Extracted Facts Finds Big Gaps","A new audit of 100 million facts GPT-4.1 believes finds accuracy far below prior benchmarks, plagued by inconsistency and hallucination.","GPT-4.1 knows a lot of things. A worrying share of them are wrong.\n\nResearchers audited GPTKB v1.5, a knowledge base built by recursively prompting GPT-4.1 to state everything it claims to know, yielding 100 million facts. Instead of relying on standard fact-completion benchmarks, which only test the queries researchers think to ask, the team built a multi-dimensional framework to sample and check that entire corpus. They compared the model's stated beliefs against established knowledge bases and found substantial divergence. The measured accuracy came in well below what prior benchmarks implied, with widespread inconsistency, ambiguity, and outright hallucination.\n\nFact-completion benchmarks are the industry's go-to way to grade a model's factual reliability, and they're built on a biased sample of facts researchers already thought to test. This audit suggests that once you look at everything a model claims to know, not just the convenient subset, the picture gets considerably worse for anyone treating LLM output as a stand-in for a database.\n\nThe paper's own prescription, pairing LLMs with neuro-symbolic verification, is itself an admission that raw model output isn't trustworthy enough to use unchecked.","[\"llm evaluation\",\"gpt-4.1\",\"knowledge bases\",\"hallucination\"]","2026-09-21T04:00:00.000Z","2026-09-21T06:53:51.890Z","2026-09-21T06:54:05.538Z","published",null,[],"https:\u002F\u002Fcdn.xyz.onl\u002Farticle-images\u002Fauditing-gpt-41s-100-million-extracted-facts-finds-big-gaps.webp","ai",[27,28,29,30],"llm evaluation","gpt-4.1","knowledge bases","hallucination",[32],{"name":33,"url":34},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2510.07024",0,{"sections":37},[38,42,46,51,56,61,66,71,76,81,86,91,96,101],{"name":39,"slug":25,"count":40,"latest_published_at":41},"AI",4175,"2026-09-21T10:30:00.000Z",{"name":43,"slug":44,"count":45,"latest_published_at":18},"Security","security",681,{"name":47,"slug":48,"count":49,"latest_published_at":50},"Policy","policy",352,"2026-09-21T10:18:06.000Z",{"name":52,"slug":53,"count":54,"latest_published_at":55},"Deals","deals",184,"2026-09-21T10:18:31.000Z",{"name":57,"slug":58,"count":59,"latest_published_at":60},"Hardware","hardware",157,"2026-09-21T11:04:12.000Z",{"name":62,"slug":63,"count":64,"latest_published_at":65},"Science","science",130,"2026-09-20T13:48:11.000Z",{"name":67,"slug":68,"count":69,"latest_published_at":70},"Consumer Tech","consumer-tech",99,"2026-09-09T17:27:33.000Z",{"name":72,"slug":73,"count":74,"latest_published_at":75},"Dev Tools","dev-tools",78,"2026-09-18T04:00:00.000Z",{"name":77,"slug":78,"count":79,"latest_published_at":80},"Software","software",75,"2026-09-10T20:41:21.000Z",{"name":82,"slug":83,"count":84,"latest_published_at":85},"Startups","startups",55,"2026-09-09T23:14:29.000Z",{"name":87,"slug":88,"count":89,"latest_published_at":90},"Gaming","gaming",43,"2026-09-10T12:18:06.000Z",{"name":92,"slug":93,"count":94,"latest_published_at":95},"General","general",42,"2026-09-18T22:35:10.000Z",{"name":97,"slug":98,"count":99,"latest_published_at":100},"Reviews","reviews",20,"2026-06-24T12:00:01.000Z",{"name":102,"slug":103,"count":104,"latest_published_at":105},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]