[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-new-benchmark-buries-ai-agents-in-75-billion-rows-of-fake-data":10,"sections":34},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":29,"feedback":33,"feedback_at":22,"cost_usd":33,"total_tokens":33},9784,"new-benchmark-buries-ai-agents-in-75-billion-rows-of-fake-data","New Benchmark Buries AI Agents in 7.5 Billion Rows of Fake Data","A new benchmark simulates a full enterprise data warehouse, and top AI agents still ace only a third of its tasks.","A new AI benchmark drops autonomous agents into a huge, fake enterprise data warehouse, and most of them get lost.\n\nResearchers built Argo-Bench, a set of 210 data science and analytics tasks modeled on a simulated food-delivery platform in New York City, complete with 81 million orders from 2024, grounded economics, and realistic fraud patterns. They exported that simulated world into an ERP warehouse of 235 tables and 7.5 billion rows, built on the Oracle E-Business Suite schema, but withheld the simulator's ground-truth state so agents have to reconstruct facts by digging through the warehouse themselves. Tasks go beyond writing SQL queries: agents must file real actions, like banning fraudulent accounts, allocating courier incentive budgets, or issuing back pay, and get graded on the consequences of those actions inside the simulator. Every task also ships with an executable reference solution, proving it is solvable using only the warehouse.\n\nThe results are a reality check for agent hype. Across 14 frontier and open-weight models, the best one scored 95 or higher on only 34.8 percent of tasks and averaged just 59.5 points. That gap matters because most text-to-SQL benchmarks test whether a model can write one correct query against a single tidy table, not whether it can navigate hundreds of linked tables and make a judgment call with real consequences, which is what actual enterprise work looks like.\n\nExisting text-to-SQL leaderboards already have a credibility problem, since audits have found their answer keys are often wrong. A benchmark that grades agents on real-world consequences, not just query syntax, is a humbling but meaningful upgrade.","[\"ai\",\"benchmarks\",\"ai-agents\",\"enterprise-data\"]","2026-10-02T04:00:00.000Z","2026-10-03T10:21:42.867Z","2026-10-03T10:21:48.327Z","published",null,[],"ai",[24,26,27,28],"benchmarks","ai-agents","enterprise-data",[30],{"name":31,"url":32},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2610.02122",0,{"sections":35},[36,39,43,47,52,56,60,65,70,75,80,85,90,95],{"name":37,"slug":24,"count":38,"latest_published_at":18},"AI",6042,{"name":40,"slug":41,"count":42,"latest_published_at":18},"Security","security",848,{"name":44,"slug":45,"count":46,"latest_published_at":18},"Policy","policy",439,{"name":48,"slug":49,"count":50,"latest_published_at":51},"Deals","deals",317,"2026-10-01T22:00:00.000Z",{"name":53,"slug":54,"count":55,"latest_published_at":18},"Hardware","hardware",199,{"name":57,"slug":58,"count":59,"latest_published_at":18},"Science","science",176,{"name":61,"slug":62,"count":63,"latest_published_at":64},"Consumer Tech","consumer-tech",155,"2026-10-01T19:54:10.000Z",{"name":66,"slug":67,"count":68,"latest_published_at":69},"Dev Tools","dev-tools",96,"2026-10-01T16:57:03.000Z",{"name":71,"slug":72,"count":73,"latest_published_at":74},"Software","software",93,"2026-09-30T21:41:11.000Z",{"name":76,"slug":77,"count":78,"latest_published_at":79},"Startups","startups",90,"2026-10-01T21:55:22.000Z",{"name":81,"slug":82,"count":83,"latest_published_at":84},"Gaming","gaming",53,"2026-10-02T02:50:39.000Z",{"name":86,"slug":87,"count":88,"latest_published_at":89},"General","general",50,"2026-09-30T21:37:54.000Z",{"name":91,"slug":92,"count":93,"latest_published_at":94},"Reviews","reviews",31,"2026-09-28T14:31:34.000Z",{"name":96,"slug":97,"count":98,"latest_published_at":99},"How-To","how-to",7,"2026-10-01T09:00:00.000Z"]