[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-agentx-runs-its-own-recommender-system-research-loop":10,"sections":34},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":29,"feedback":33,"feedback_at":22,"cost_usd":33,"total_tokens":33},7721,"agentx-runs-its-own-recommender-system-research-loop","AgentX Runs Its Own Recommender System Research Loop","A dual-agent AI system now proposes, runs, and diagnoses its own recommender-model experiments, with most beating baseline accuracy in production.","An AI research system is now largely running its own recommender-model experiments, and it works often enough to matter.\n\nAgentX-Model pairs two agents: a Research Agent that turns papers and past results into proposals, and a Model Agent that runs multi-round experiments and reports back code, measurements, and open questions. The Research Agent then picks a starting point and sets the next question, so each experiment builds on the last. Work is organized into four moves: Reproduce, Follow-up, Composition, and Diagnose, with the last used to chase down problems like prediction bias flagged by business feedback. In production, 560 of 636 completed experiments beat their baseline AUC, and some eventually outperformed every prior version in their lineage.\n\nThe real test is the online results: the five most recent A\u002FB evaluations reported 10-15% gains in acquisition efficiency, 15-20% gains in target-segment ad spend, and 0.3-0.8% gains in watch time, the last using about 10% fewer FLOPs and parameters. That last number is the interesting one - the system didn't just chase accuracy, it found a cheaper model that did as well or better. A separate benchmark also found that fancier experiment-scheduling didn't help once the agents were already picking sensible candidates on their own, a useful data point against over-engineering the orchestration layer.\n\nIt's a single team's self-reported numbers in a preprint, not an independently audited result, and the paper doesn't say which company's recommender stack this is running on. Still, if automated research loops like this hold up outside one lab, a lot of applied ML teams are about to get quietly smaller.","[\"ai\",\"recommender-systems\",\"ai-agents\",\"machine-learning\"]","2026-09-25T04:00:00.000Z","2026-09-25T06:28:06.510Z","2026-09-25T06:28:11.868Z","published",null,[],"ai",[24,26,27,28],"recommender-systems","ai-agents","machine-learning",[30],{"name":31,"url":32},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2609.30001",0,{"sections":35},[36,39,44,49,54,59,64,69,74,79,84,89,93,98],{"name":37,"slug":24,"count":38,"latest_published_at":18},"AI",4466,{"name":40,"slug":41,"count":42,"latest_published_at":43},"Security","security",729,"2026-09-24T19:54:21.000Z",{"name":45,"slug":46,"count":47,"latest_published_at":48},"Policy","policy",386,"2026-09-24T23:50:55.000Z",{"name":50,"slug":51,"count":52,"latest_published_at":53},"Deals","deals",237,"2026-09-24T22:00:00.000Z",{"name":55,"slug":56,"count":57,"latest_published_at":58},"Hardware","hardware",182,"2026-09-25T01:25:53.000Z",{"name":60,"slug":61,"count":62,"latest_published_at":63},"Science","science",138,"2026-09-24T18:24:52.000Z",{"name":65,"slug":66,"count":67,"latest_published_at":68},"Consumer Tech","consumer-tech",128,"2026-09-24T19:24:34.000Z",{"name":70,"slug":71,"count":72,"latest_published_at":73},"Software","software",88,"2026-09-24T23:06:55.000Z",{"name":75,"slug":76,"count":77,"latest_published_at":78},"Dev Tools","dev-tools",79,"2026-09-22T22:21:13.000Z",{"name":80,"slug":81,"count":82,"latest_published_at":83},"Startups","startups",71,"2026-09-24T20:45:00.000Z",{"name":85,"slug":86,"count":87,"latest_published_at":88},"Gaming","gaming",46,"2026-09-24T17:52:29.000Z",{"name":90,"slug":91,"count":87,"latest_published_at":92},"General","general","2026-09-25T02:12:57.000Z",{"name":94,"slug":95,"count":96,"latest_published_at":97},"Reviews","reviews",30,"2026-09-24T20:07:31.000Z",{"name":99,"slug":100,"count":101,"latest_published_at":102},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]