[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-new-verifier-stops-llms-from-contradicting-retracted-claims":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},6360,"new-verifier-stops-llms-from-contradicting-retracted-claims","New Verifier Stops LLMs From Contradicting Retracted Claims","A linear-time runtime check tracks what a conversation has established and flags LLM replies that lean on claims already retracted.","A new runtime check aims to stop LLMs from confidently repeating claims a conversation already retracted.\n\nThe paper, posted to arXiv, describes a verifier that sits between an LLM and its own chat history: an Interpreter model classifies each conversational turn into one of eight epistemic operations, and a symbolic engine files those into a dependency map showing what every claim rests on and whether that support still holds. Checking whether a new reply is grounded becomes a linear-time walk over that map, no extra LLM call required, and when a premise gets retracted the conflict tracking propagates through the map to flag exactly which downstream conclusions lose their footing. On two third-party benchmarks built around superseded premises, ReviseQA and MemoryAgentBench's fact-consolidation split, the verifier beat a budget-matched retrieval baseline across five QA models and pushed MemoryAgentBench single-hop accuracy from a range of 0.46-0.95 up to 0.93-0.98. With the verifier attached, even a 7B model outperformed an unassisted GPT-4o.\n\nThe interesting part is cost, not accuracy. Full conversation context can run to 114,000 tokens; this system keeps prompts near 800 tokens regardless of conversation length, and checks a retraction in under a microsecond at 2,000 turns. That overhead is low enough to run on every single turn of a production agent, which matters because context-manipulation attacks - feeding an agent a plausible continuation built on premises it already abandoned - are a live exploit against deployed systems, not a hypothetical one.\n\nOne caveat: these results mostly rely on the benchmarks' own structured updates. When a GPT-4o Interpreter had to extract those updates from raw text itself, accuracy barely moved - which is the part that will actually decide whether this holds up outside a lab.","[\"llm\",\"ai-agents\",\"benchmarks\",\"ai-safety\"]","2026-09-11T04:00:00.000Z","2026-09-11T08:25:24.233Z","2026-09-11T08:25:36.056Z","published",null,[],"ai",[26,27,28,29],"llm","ai-agents","benchmarks","ai-safety",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2605.14175",0,{"sections":36},[37,40,44,48,53,58,63,66,71,75,80,85,90,95],{"name":38,"slug":24,"count":39,"latest_published_at":18},"AI",3543,{"name":41,"slug":42,"count":43,"latest_published_at":18},"Security","security",637,{"name":45,"slug":46,"count":47,"latest_published_at":18},"Policy","policy",338,{"name":49,"slug":50,"count":51,"latest_published_at":52},"Deals","deals",179,"2026-06-29T20:02:07.000Z",{"name":54,"slug":55,"count":56,"latest_published_at":57},"Hardware","hardware",153,"2026-09-09T15:12:32.000Z",{"name":59,"slug":60,"count":61,"latest_published_at":62},"Consumer Tech","consumer-tech",99,"2026-09-09T17:27:33.000Z",{"name":64,"slug":65,"count":61,"latest_published_at":18},"Science","science",{"name":67,"slug":68,"count":69,"latest_published_at":70},"Software","software",75,"2026-09-10T20:41:21.000Z",{"name":72,"slug":73,"count":74,"latest_published_at":18},"Dev Tools","dev-tools",70,{"name":76,"slug":77,"count":78,"latest_published_at":79},"Startups","startups",55,"2026-09-09T23:14:29.000Z",{"name":81,"slug":82,"count":83,"latest_published_at":84},"Gaming","gaming",43,"2026-09-10T12:18:06.000Z",{"name":86,"slug":87,"count":88,"latest_published_at":89},"General","general",41,"2026-09-08T01:57:23.000Z",{"name":91,"slug":92,"count":93,"latest_published_at":94},"Reviews","reviews",20,"2026-06-24T12:00:01.000Z",{"name":96,"slug":97,"count":98,"latest_published_at":99},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]