[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-study-finds-mobile-ai-agents-stumble-on-personalized-phone-screens":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},10600,"study-finds-mobile-ai-agents-stumble-on-personalized-phone-screens","Study finds mobile AI agents stumble on personalized phone screens","A new benchmark shows AI agents that control your phone perform worse once your home screen, apps, and files look like yours instead of a generic test account.","AI agents that tap and swipe through phone apps on your behalf get noticeably worse once the phone looks like yours.\n\nResearchers built a tool called PAIR that recreates personalized app states - the exact home screens, saved items, and content a real user would see - so the same task can be tested across many different simulated users. Running six existing mobile GUI agents through this setup, they found consistent drops in task success once interfaces reflected individual user histories, with subgoal achievement falling 6.98 to 15.4 percentage points compared to generic, unpersonalized test environments. The gap widened further, to 8.77-22.0 percentage points, when the target was something from a user's own content rather than a neutral stand-in. Most failures happened when an agent picked a different, similar-looking item instead of the one it was actually told to find, especially before the correct target had appeared on screen.\n\nThe researchers also trained a new method, RePAIR, that learns directly from these cross-user differences in how subgoals succeed or fail. On users it had never seen during training, RePAIR beat its supervised fine-tuning starting point by 5.87 percentage points on user-conditioned subgoal accuracy, 7.50 points on getting every step right, and 9.42 points on overall task success.\n\nThe gap matters because most agent benchmarks run on tidy demo accounts, not the clutter of someone's actual phone. If personalization alone costs an agent 7 to 22 percentage points of reliability, the numbers a company shows off are not the numbers a real user should expect.\n\nLab success and real phone success are apparently two different benchmarks. For now, only one of them gets measured and marketed.","[\"ai agents\",\"gui agents\",\"personalization\",\"benchmarks\"]","2026-10-07T04:00:00.000Z","2026-10-08T22:04:51.501Z","2026-10-08T22:04:56.010Z","published",null,[],"ai",[26,27,28,29],"ai agents","gui agents","personalization","benchmarks",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2610.07972",0,{"sections":36},[37,41,46,51,56,61,66,71,76,80,85,90,95,100],{"name":38,"slug":24,"count":39,"latest_published_at":40},"AI",6448,"2026-10-07T18:45:00.000Z",{"name":42,"slug":43,"count":44,"latest_published_at":45},"Security","security",904,"2026-10-07T19:53:42.000Z",{"name":47,"slug":48,"count":49,"latest_published_at":50},"Policy","policy",474,"2026-10-07T18:23:21.000Z",{"name":52,"slug":53,"count":54,"latest_published_at":55},"Deals","deals",453,"2026-10-07T23:58:31.000Z",{"name":57,"slug":58,"count":59,"latest_published_at":60},"Hardware","hardware",222,"2026-10-07T21:19:54.000Z",{"name":62,"slug":63,"count":64,"latest_published_at":65},"Science","science",186,"2026-10-06T21:20:39.000Z",{"name":67,"slug":68,"count":69,"latest_published_at":70},"Consumer Tech","consumer-tech",174,"2026-10-07T17:41:41.000Z",{"name":72,"slug":73,"count":74,"latest_published_at":75},"Software","software",113,"2026-10-07T18:10:00.000Z",{"name":77,"slug":78,"count":74,"latest_published_at":79},"Startups","startups","2026-10-07T23:36:57.000Z",{"name":81,"slug":82,"count":83,"latest_published_at":84},"Dev Tools","dev-tools",105,"2026-10-07T16:59:11.000Z",{"name":86,"slug":87,"count":88,"latest_published_at":89},"General","general",61,"2026-10-07T22:00:24.000Z",{"name":91,"slug":92,"count":93,"latest_published_at":94},"Gaming","gaming",56,"2026-10-07T12:00:00.000Z",{"name":96,"slug":97,"count":98,"latest_published_at":99},"Reviews","reviews",33,"2026-10-05T11:57:17.000Z",{"name":101,"slug":102,"count":103,"latest_published_at":104},"How-To","how-to",8,"2026-10-05T09:00:00.000Z"]