[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-opengameeval-tests-ai-coding-agents-inside-roblox-studio":10,"sections":34},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":29,"feedback":33,"feedback_at":22,"cost_usd":33,"total_tokens":33},9997,"opengameeval-tests-ai-coding-agents-inside-roblox-studio","OpenGameEval tests AI coding agents inside Roblox Studio","A new benchmark finds top AI coding agents solve barely half of Roblox Studio tasks on a first try, and exploration habits predict success.","The best AI coding agents still fail roughly half of basic Roblox Studio tasks on their first attempt, according to a new benchmark.\n\nResearchers built OpenGameEval, a framework that runs language models as coding agents inside live Roblox Studio sessions and grades them with automated checks on both the edited scene and a simulated play session. They tested 13 frontier models against 84 human-curated tasks, giving each model 16 attempts per task. The top-performing model solved 51.7% of tasks on a single try, but that dropped to 39.4% when it had to succeed five times out of five. Six tasks went unsolved by every model tested.\n\nThe benchmark's useful trick is splitting \"look around\" tools from \"make a change\" tools, so researchers can measure exploration separately from final success. That split shows models that inspected every object a reference solution touched passed up to 13.4 percentage points more often than models that skipped the look-around step entirely. The failures, in other words, often start before any code gets written.\n\nMost agentic coding leaderboards only care whether a model reaches the right answer, however it gets there. OpenGameEval's exploration-tracking approach is a reminder that knowing where to look may matter as much as knowing what to type.","[\"ai\",\"benchmarks\",\"roblox\",\"agentic-coding\"]","2026-10-05T04:00:00.000Z","2026-10-05T17:16:03.266Z","2026-10-05T17:16:09.262Z","published",null,[],"ai",[24,26,27,28],"benchmarks","roblox","agentic-coding",[30],{"name":31,"url":32},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2610.02563",0,{"sections":35},[36,39,43,48,53,58,62,67,71,75,80,85,90,95],{"name":37,"slug":24,"count":38,"latest_published_at":18},"AI",6233,{"name":40,"slug":41,"count":42,"latest_published_at":18},"Security","security",868,{"name":44,"slug":45,"count":46,"latest_published_at":47},"Policy","policy",444,"2026-10-03T15:02:01.000Z",{"name":49,"slug":50,"count":51,"latest_published_at":52},"Deals","deals",323,"2026-10-04T13:00:00.000Z",{"name":54,"slug":55,"count":56,"latest_published_at":57},"Hardware","hardware",204,"2026-10-03T14:50:50.000Z",{"name":59,"slug":60,"count":61,"latest_published_at":18},"Science","science",177,{"name":63,"slug":64,"count":65,"latest_published_at":66},"Consumer Tech","consumer-tech",158,"2026-10-03T03:21:12.000Z",{"name":68,"slug":69,"count":70,"latest_published_at":18},"Dev Tools","dev-tools",97,{"name":72,"slug":73,"count":70,"latest_published_at":74},"Software","software","2026-10-04T10:00:00.000Z",{"name":76,"slug":77,"count":78,"latest_published_at":79},"Startups","startups",92,"2026-10-04T14:36:25.000Z",{"name":81,"slug":82,"count":83,"latest_published_at":84},"Gaming","gaming",53,"2026-10-02T02:50:39.000Z",{"name":86,"slug":87,"count":88,"latest_published_at":89},"General","general",51,"2026-10-05T02:35:01.000Z",{"name":91,"slug":92,"count":93,"latest_published_at":94},"Reviews","reviews",32,"2026-10-02T18:00:00.000Z",{"name":96,"slug":97,"count":98,"latest_published_at":99},"How-To","how-to",7,"2026-10-01T09:00:00.000Z"]