[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-study-finds-ai-assistants-often-fake-remembering-past-chats":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},6653,"study-finds-ai-assistants-often-fake-remembering-past-chats","Study Finds AI Assistants Often Fake Remembering Past Chats","MIRAGE, a new benchmark, tests whether multimodal AI agents truly retrieve old evidence or just guess once their conversation history compresses.","A new benchmark suggests AI personal assistants often can't tell the difference between remembering something and making it up.\n\nResearchers built MIRAGE (Multimodal Interaction Retrieval, Attribution, and Grounding Evaluation) to test how well multimodal AI agents use evidence from earlier in a conversation, including files, chat history, and workspace state, while holding the underlying facts and questions constant. The only thing they varied was conversation state: how deep into a chat the question landed, and whether it came before or after the conversation got compressed into a summary. Seven frontier and open-weight models were checked on three things: could they tell if a question was even answerable from prior evidence, could they find the right source, and could they answer from it. The failures did not follow one steady decline. The team found two separate, unpredictable failure patterns, one tied to how deep a conversation went before compression and another tied to what happened after.\n\nThat distinction matters because most agent benchmarks only grade the final answer, which lets a good guess look identical to genuine recall. Open-weight models were especially prone to this: they leaned on raw context and rarely switched to re-retrieving evidence through a tool once that context's origin got shaky. For products sold on the promise of a personal assistant that remembers your files and past conversations, that is a meaningful gap between marketing and mechanism.\n\nAn assistant that sounds sure and an assistant that actually checked its notes produce the same sentence, until someone builds a test that can tell them apart.","[\"ai agents\",\"llm evaluation\",\"multimodal ai\",\"ai benchmarks\"]","2026-09-17T04:00:00.000Z","2026-09-18T04:30:12.957Z","2026-09-18T04:30:24.869Z","published",null,[],"ai",[26,27,28,29],"ai agents","llm evaluation","multimodal ai","ai benchmarks",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2609.19059",0,{"sections":36},[37,41,45,50,55,59,63,68,73,77,82,87,92,97],{"name":38,"slug":24,"count":39,"latest_published_at":40},"AI",3853,"2026-09-17T08:27:09.000Z",{"name":42,"slug":43,"count":44,"latest_published_at":18},"Security","security",648,{"name":46,"slug":47,"count":48,"latest_published_at":49},"Policy","policy",338,"2026-09-11T04:00:00.000Z",{"name":51,"slug":52,"count":53,"latest_published_at":54},"Deals","deals",179,"2026-06-29T20:02:07.000Z",{"name":56,"slug":57,"count":58,"latest_published_at":18},"Hardware","hardware",154,{"name":60,"slug":61,"count":62,"latest_published_at":18},"Science","science",114,{"name":64,"slug":65,"count":66,"latest_published_at":67},"Consumer Tech","consumer-tech",99,"2026-09-09T17:27:33.000Z",{"name":69,"slug":70,"count":71,"latest_published_at":72},"Software","software",75,"2026-09-10T20:41:21.000Z",{"name":74,"slug":75,"count":76,"latest_published_at":18},"Dev Tools","dev-tools",73,{"name":78,"slug":79,"count":80,"latest_published_at":81},"Startups","startups",55,"2026-09-09T23:14:29.000Z",{"name":83,"slug":84,"count":85,"latest_published_at":86},"Gaming","gaming",43,"2026-09-10T12:18:06.000Z",{"name":88,"slug":89,"count":90,"latest_published_at":91},"General","general",41,"2026-09-08T01:57:23.000Z",{"name":93,"slug":94,"count":95,"latest_published_at":96},"Reviews","reviews",20,"2026-06-24T12:00:01.000Z",{"name":98,"slug":99,"count":100,"latest_published_at":101},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]