[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-new-benchmark-shows-ai-agents-fail-at-long-term-habit-tracking":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},6769,"new-benchmark-shows-ai-agents-fail-at-long-term-habit-tracking","New Benchmark Shows AI Agents Fail at Long Term Habit Tracking","SimLife tests whether AI can learn household routines over weeks of simulated life, and most models default to guesswork instead of real reasoning.","A new benchmark called SimLife-BP tests something most AI evaluations skip entirely: can a model actually learn your habits over time, or is it just guessing based on what happened most recently?\n\nResearchers built SimLife, a simulation platform that generates long-term household life data, including visual observations, action logs, and synthetic dialogue with audio. From that, they built SimLife-BP, a benchmark of 106 episodes averaging over 15 hours and nearly 39 in-game days each, totaling 1,439 question-answer pairs. The questions probe whether models can infer the actual rules behind a household's routines, not just predict the next likely action, testing direct, counterfactual, noisy, and inverse reasoning. The results were not flattering. Frontier models mostly leaned on frequency-based shortcuts, guessing based on what happened most often, rather than reasoning through if-then logic, and they struggled badly when routines changed partway through.\n\nThat gap matters more than it sounds. Every pitch for a household robot, a personal assistant, or a long-running agentic tool assumes the system will get better at understanding you the longer it watches. This benchmark suggests that assumption is shakier than the marketing implies: models can mimic pattern recognition over short windows but lose the thread once a routine shifts or noise enters the picture.\n\nIt is also a useful corrective to the current AI narrative. Most benchmarks reward short-context tasks where memorizing surface patterns looks like understanding. SimLife-BP is built to catch that difference, and frontier models mostly failed to clear it. Worth remembering next time a company demos an agent that promises to \"learn your routine.\"","[\"ai-benchmarks\",\"embodied-ai\",\"agents\",\"arxiv\"]","2026-09-18T04:00:00.000Z","2026-09-18T15:49:53.172Z","2026-09-18T15:50:05.147Z","published",null,[],"ai",[26,27,28,29],"ai-benchmarks","embodied-ai","agents","arxiv",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2609.19610",0,{"sections":36},[37,40,44,49,54,58,62,67,72,77,82,87,92,97],{"name":38,"slug":24,"count":39,"latest_published_at":18},"AI",3959,{"name":41,"slug":42,"count":43,"latest_published_at":18},"Security","security",652,{"name":45,"slug":46,"count":47,"latest_published_at":48},"Policy","policy",338,"2026-09-11T04:00:00.000Z",{"name":50,"slug":51,"count":52,"latest_published_at":53},"Deals","deals",179,"2026-06-29T20:02:07.000Z",{"name":55,"slug":56,"count":57,"latest_published_at":18},"Hardware","hardware",155,{"name":59,"slug":60,"count":61,"latest_published_at":18},"Science","science",116,{"name":63,"slug":64,"count":65,"latest_published_at":66},"Consumer Tech","consumer-tech",99,"2026-09-09T17:27:33.000Z",{"name":68,"slug":69,"count":70,"latest_published_at":71},"Dev Tools","dev-tools",76,"2026-09-18T01:04:54.000Z",{"name":73,"slug":74,"count":75,"latest_published_at":76},"Software","software",75,"2026-09-10T20:41:21.000Z",{"name":78,"slug":79,"count":80,"latest_published_at":81},"Startups","startups",55,"2026-09-09T23:14:29.000Z",{"name":83,"slug":84,"count":85,"latest_published_at":86},"Gaming","gaming",43,"2026-09-10T12:18:06.000Z",{"name":88,"slug":89,"count":90,"latest_published_at":91},"General","general",41,"2026-09-08T01:57:23.000Z",{"name":93,"slug":94,"count":95,"latest_published_at":96},"Reviews","reviews",20,"2026-06-24T12:00:01.000Z",{"name":98,"slug":99,"count":100,"latest_published_at":101},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]