[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-ai-agents-learn-to-judge-actions-not-just-copy-experts":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},11078,"ai-agents-learn-to-judge-actions-not-just-copy-experts","AI Agents Learn to Judge Actions, Not Just Copy Experts","A new reinforcement learning technique trains AI agents to pick expert actions over plausible mistakes, improving both agent tasks and math benchmarks.","A new training method teaches AI agents to spot a bad move instead of just copying a good one.\n\nResearchers built Agentic Critical Training (ACT), which uses reinforcement learning with verifiable rewards to train a model to pick the expert's action over a plausible but wrong alternative, rather than just imitating a transcript. At each step of a demonstration, the model sees the expert action and a decoy sampled from its own earlier policy, shown in random order, and only gets rewarded for choosing correctly. It has to produce its own reasoning to get there - there's no scripted rationale to memorize. The team tested two models, Qwen3-8B and Olmo-3-7B-Instruct, as a warm-up step before standard imitation learning and reinforcement learning, across three agent benchmarks: ALFWorld, WebShop, and ScienceWorld.\n\nThe gains are not trivial. ACT added an average of 5.85 points over plain imitation learning and 4.12 points over the usual imitation-then-reinforcement-learning pipeline without it, and on Olmo's ScienceWorld runs the full pipeline beat basic chain-of-thought prompting by 15.36 points. It also improved results on scenarios the models had not seen during training, and on math and science reasoning benchmarks the models were never specifically tuned for - a sign that learning to judge actions teaches something more general than learning to generate them.\n\nThe researchers also checked whether the gains were just an artifact of extra training time or data, and found they were not - the improvement came from the comparison task itself. That is a more rigorous control than most agent-training papers bother with, and it is the detail that actually makes the numbers worth trusting.","[\"reinforcement-learning\",\"ai-training\",\"language-models\",\"agentic-ai\"]","2026-10-09T04:00:00.000Z","2026-10-10T05:05:37.272Z","2026-10-10T05:05:39.149Z","published",null,[],"ai",[26,27,28,29],"reinforcement-learning","ai-training","language-models","agentic-ai",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2603.08706",0,{"sections":36},[37,40,44,49,54,58,62,67,72,77,81,86,91,96],{"name":38,"slug":24,"count":39,"latest_published_at":18},"AI",6804,{"name":41,"slug":42,"count":43,"latest_published_at":18},"Security","security",934,{"name":45,"slug":46,"count":47,"latest_published_at":48},"Policy","policy",486,"2026-10-08T22:40:11.000Z",{"name":50,"slug":51,"count":52,"latest_published_at":53},"Deals","deals",474,"2026-10-08T22:00:00.000Z",{"name":55,"slug":56,"count":57,"latest_published_at":18},"Hardware","hardware",232,{"name":59,"slug":60,"count":61,"latest_published_at":18},"Science","science",194,{"name":63,"slug":64,"count":65,"latest_published_at":66},"Consumer Tech","consumer-tech",181,"2026-10-08T23:26:35.000Z",{"name":68,"slug":69,"count":70,"latest_published_at":71},"Startups","startups",117,"2026-10-08T16:45:00.000Z",{"name":73,"slug":74,"count":75,"latest_published_at":76},"Software","software",114,"2026-10-08T17:57:01.000Z",{"name":78,"slug":79,"count":80,"latest_published_at":18},"Dev Tools","dev-tools",106,{"name":82,"slug":83,"count":84,"latest_published_at":85},"General","general",66,"2026-10-09T04:46:11.000Z",{"name":87,"slug":88,"count":89,"latest_published_at":90},"Gaming","gaming",58,"2026-10-08T20:08:45.000Z",{"name":92,"slug":93,"count":94,"latest_published_at":95},"Reviews","reviews",34,"2026-10-08T14:00:22.000Z",{"name":97,"slug":98,"count":99,"latest_published_at":100},"How-To","how-to",8,"2026-10-05T09:00:00.000Z"]