[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-agent-benchmarks-hide-a-runtime-trap-study-finds":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},9224,"agent-benchmarks-hide-a-runtime-trap-study-finds","Agent Benchmarks Hide a Runtime Trap, Study Finds","Researchers found Qwen3-8B coding agents lose up to 90 percent of their performance when deployed on a different runtime than they were trained on.","Researchers just showed that AI coding agents can collapse in quality when moved to a different software environment, even though nothing about the underlying model changed.\n\nThe study trained separate Qwen3-8B agents on two types of execution environments: persistent runtimes, which keep Python variables between actions, and stateless runtimes, which wipe them clean after each turn. Agents learned task-specific tricks for storing progress depending on which setup they trained in. When researchers deployed a persistent-trained agent into a stateless environment and capped it at 80 tool calls per turn, it scored only 0.61 of the optimal outcome, versus 0.77 when kept on its native runtime, while burning 4.4 times more tokens rebuilding its working notes. Tighten that cap to 25 calls per turn and the mismatched agent's score crashes to 0.07, while the matched agent still hits 0.66. The team confirmed the pattern across three tasks, multiple training seeds, a second rollout, and two additional base models.\n\nThis matters because most agent benchmarks test a single, fixed runtime and quietly assume the result holds wherever the agent gets deployed. That assumption echoes an old machine-learning trap: models tuned on clean conditions that fall apart once shipped into messier production setups. Here the damage shows up as either wasted compute or near-total task failure, and a leaderboard score alone cannot tell you which one you are about to get.\n\nCall it the shipping-container problem: an agent that packs its cargo one way falls apart the moment someone swaps the truck.","[\"ai-agents\",\"benchmarking\",\"llm-research\",\"arxiv\"]","2026-10-01T04:00:00.000Z","2026-10-02T02:46:40.243Z","2026-10-02T02:46:42.552Z","published",null,[],"ai",[26,27,28,29],"ai-agents","benchmarking","llm-research","arxiv",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2603.01209",0,{"sections":36},[37,40,44,48,53,58,62,67,72,76,81,86,91,96],{"name":38,"slug":24,"count":39,"latest_published_at":18},"AI",5629,{"name":41,"slug":42,"count":43,"latest_published_at":18},"Security","security",816,{"name":45,"slug":46,"count":47,"latest_published_at":18},"Policy","policy",430,{"name":49,"slug":50,"count":51,"latest_published_at":52},"Deals","deals",298,"2026-09-30T21:00:26.000Z",{"name":54,"slug":55,"count":56,"latest_published_at":57},"Hardware","hardware",196,"2026-09-30T13:00:00.000Z",{"name":59,"slug":60,"count":61,"latest_published_at":18},"Science","science",163,{"name":63,"slug":64,"count":65,"latest_published_at":66},"Consumer Tech","consumer-tech",149,"2026-09-30T22:57:11.000Z",{"name":68,"slug":69,"count":70,"latest_published_at":71},"Dev Tools","dev-tools",93,"2026-10-01T02:30:48.000Z",{"name":73,"slug":74,"count":70,"latest_published_at":75},"Software","software","2026-09-30T21:41:11.000Z",{"name":77,"slug":78,"count":79,"latest_published_at":80},"Startups","startups",84,"2026-09-30T20:39:09.000Z",{"name":82,"slug":83,"count":84,"latest_published_at":85},"Gaming","gaming",51,"2026-09-30T16:24:30.000Z",{"name":87,"slug":88,"count":89,"latest_published_at":90},"General","general",50,"2026-09-30T21:37:54.000Z",{"name":92,"slug":93,"count":94,"latest_published_at":95},"Reviews","reviews",31,"2026-09-28T14:31:34.000Z",{"name":97,"slug":98,"count":99,"latest_published_at":100},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]