[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-new-training-method-lets-ai-agents-rehearse-actions-internally":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},9238,"new-training-method-lets-ai-agents-rehearse-actions-internally","New Training Method Lets AI Agents Rehearse Actions Internally","EnvACE trains AI agents to simulate their own tool responses during training, cutting reliance on costly external environments without sacrificing performance.","A new reinforcement learning method called EnvACE trains AI agents without ever touching a real tool or server.\n\nResearchers built EnvACE as an alternative to the usual way of training large language model agents for multi-step tool use. Instead of letting the model call real APIs or synthetic test environments, the policy alternates between two roles: it generates a tool call, then plays the environment itself, predicting what response that call would produce. Both roles are optimized together using rewards for whether the overall task succeeds. Across four benchmarks, BFCL-v4, tau^2-Bench, VitaBench, and FinMCP-Bench, the approach beat baselines that scale up by adding more real environments.\n\nBuilding and verifying executable test environments for agent training is expensive and slow, and external simulators are often a poor match for real-world systems. EnvACE's bet is that an agent can learn a usable internal model of 'if I do X, the environment will likely say Y' well enough to skip that infrastructure entirely. The same internalized model lets the agent privately rehearse a move before committing to it at deployment time, with the paper reporting further gains from a moderate amount of this rehearsal.\n\nSelf-play training where a model simulates both the actor and the world it acts in is a neat trick, but the agent's sense of 'the environment' is only as good as its own imagination, a wrinkle that matters more for a financial tool-use benchmark like FinMCP-Bench than for a coding sandbox where mistakes are cheap.","[\"ai agents\",\"reinforcement learning\",\"llm benchmarks\",\"world models\"]","2026-10-01T04:00:00.000Z","2026-10-02T03:26:31.422Z","2026-10-02T03:26:37.029Z","published",null,[],"ai",[26,27,28,29],"ai agents","reinforcement learning","llm benchmarks","world models",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2608.06197",0,{"sections":36},[37,41,45,49,54,58,62,67,72,77,82,87,92,97],{"name":38,"slug":24,"count":39,"latest_published_at":40},"AI",6014,"2026-10-02T04:00:00.000Z",{"name":42,"slug":43,"count":44,"latest_published_at":40},"Security","security",847,{"name":46,"slug":47,"count":48,"latest_published_at":40},"Policy","policy",439,{"name":50,"slug":51,"count":52,"latest_published_at":53},"Deals","deals",317,"2026-10-01T22:00:00.000Z",{"name":55,"slug":56,"count":57,"latest_published_at":40},"Hardware","hardware",199,{"name":59,"slug":60,"count":61,"latest_published_at":40},"Science","science",176,{"name":63,"slug":64,"count":65,"latest_published_at":66},"Consumer Tech","consumer-tech",155,"2026-10-01T19:54:10.000Z",{"name":68,"slug":69,"count":70,"latest_published_at":71},"Dev Tools","dev-tools",96,"2026-10-01T16:57:03.000Z",{"name":73,"slug":74,"count":75,"latest_published_at":76},"Software","software",93,"2026-09-30T21:41:11.000Z",{"name":78,"slug":79,"count":80,"latest_published_at":81},"Startups","startups",90,"2026-10-01T21:55:22.000Z",{"name":83,"slug":84,"count":85,"latest_published_at":86},"Gaming","gaming",53,"2026-10-02T02:50:39.000Z",{"name":88,"slug":89,"count":90,"latest_published_at":91},"General","general",50,"2026-09-30T21:37:54.000Z",{"name":93,"slug":94,"count":95,"latest_published_at":96},"Reviews","reviews",31,"2026-09-28T14:31:34.000Z",{"name":98,"slug":99,"count":100,"latest_published_at":101},"How-To","how-to",7,"2026-10-01T09:00:00.000Z"]