[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-researchers-catch-ai-agents-cheating-at-database-questions":10,"sections":34},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":29,"feedback":33,"feedback_at":22,"cost_usd":33,"total_tokens":33},10956,"researchers-catch-ai-agents-cheating-at-database-questions","Researchers Catch AI Agents Cheating at Database Questions","A new benchmark called AmbiTab reveals how AI agents answering database questions often get help they shouldn't from test simulators, skewing results.","AI agents that answer questions about spreadsheets and databases have a cheating problem, and a new paper tries to measure it.\n\nResearchers built a benchmark called AmbiTab to test how well AI agents handle vague questions about tabular data, the kind of mid-conversation back and forth needed before a bot can write a correct SQL query. The paper defines what it calls an \"ambiguous verifiable task,\" splitting an agent into two parts: one that asks clarifying questions and one that generates the actual answer. It also formalizes \"oracle leakage,\" meaning cases where the simulated user in a test accidentally reveals more than a real person would. The team combined six existing ambiguous text-to-SQL datasets into one shared format, then trained a question-asking policy with reinforcement learning, which improved disambiguation scores across all six datasets and overall task success on five of them.\n\nThis matters because most chatbot benchmarks conflate two different skills: knowing what to ask, and knowing how to answer. If a test's simulated user drops hints a real customer never would, an agent looks sharper than it is in production. That gap is exactly why so many enterprise text-to-SQL and data-assistant tools work in demos and stumble on real tickets, where nobody phrases a request as cleanly as a benchmark expects.\n\nIt's a useful correction, not a breakthrough. Benchmarks that quietly reward good test-taking rather than good reasoning are an old problem in machine learning, and this is one more sign builders should read the eval's fine print before trusting the leaderboard.","[\"ai\",\"text-to-sql\",\"benchmarks\",\"llm-agents\"]","2026-10-09T04:00:00.000Z","2026-10-09T23:18:06.615Z","2026-10-09T23:18:12.139Z","published",null,[],"ai",[24,26,27,28],"text-to-sql","benchmarks","llm-agents",[30],{"name":31,"url":32},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2610.10740",0,{"sections":35},[36,39,43,48,53,57,61,66,71,76,81,86,91,96],{"name":37,"slug":24,"count":38,"latest_published_at":18},"AI",6708,{"name":40,"slug":41,"count":42,"latest_published_at":18},"Security","security",931,{"name":44,"slug":45,"count":46,"latest_published_at":47},"Policy","policy",486,"2026-10-08T22:40:11.000Z",{"name":49,"slug":50,"count":51,"latest_published_at":52},"Deals","deals",474,"2026-10-08T22:00:00.000Z",{"name":54,"slug":55,"count":56,"latest_published_at":18},"Hardware","hardware",231,{"name":58,"slug":59,"count":60,"latest_published_at":18},"Science","science",192,{"name":62,"slug":63,"count":64,"latest_published_at":65},"Consumer Tech","consumer-tech",181,"2026-10-08T23:26:35.000Z",{"name":67,"slug":68,"count":69,"latest_published_at":70},"Startups","startups",117,"2026-10-08T16:45:00.000Z",{"name":72,"slug":73,"count":74,"latest_published_at":75},"Software","software",114,"2026-10-08T17:57:01.000Z",{"name":77,"slug":78,"count":79,"latest_published_at":80},"Dev Tools","dev-tools",105,"2026-10-07T16:59:11.000Z",{"name":82,"slug":83,"count":84,"latest_published_at":85},"General","general",66,"2026-10-09T04:46:11.000Z",{"name":87,"slug":88,"count":89,"latest_published_at":90},"Gaming","gaming",58,"2026-10-08T20:08:45.000Z",{"name":92,"slug":93,"count":94,"latest_published_at":95},"Reviews","reviews",34,"2026-10-08T14:00:22.000Z",{"name":97,"slug":98,"count":99,"latest_published_at":100},"How-To","how-to",8,"2026-10-05T09:00:00.000Z"]