[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-new-benchmark-tests-whether-ai-coding-agents-actually-learn-skills":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},9130,"new-benchmark-tests-whether-ai-coding-agents-actually-learn-skills","New benchmark tests whether AI coding agents actually learn skills","EngramBench shows that reusable skill memories cut AI coding agents' task time by over 55 percent, though they still can't replace raw reasoning power.","Researchers built a benchmark to find out if AI coding agents are actually learning skills, or just remembering old answers.\n\nThe new benchmark, called EngramBench, gives AI agents 30 training tasks and then 13 brand-new transfer tasks the agents have never seen before. The design rules out an easy shortcut: every task shares underlying skills with ones the agent already solved, but no task shares actual code, so a model can't just recall a near-identical snippet from training data and call it learning. The team ran 48 separate multi-hour test sessions, each driven by a simulated user interacting with the agent throughout, and checked the results against human-expert review.\n\nThe headline result is a split verdict. Storing a bank of learned skills for an agent to draw on doesn't make its code any more correct - that ceiling is still set by the underlying model's own reasoning limits. But it does make agents dramatically more efficient, cutting coding time by more than 55 percent by steering them away from repeated, token-burning trial and error.\n\nIn other words, skill memory isn't a brain upgrade. It's a better map for a brain that was going to get there eventually, just more slowly and more expensively.","[\"ai agents\",\"benchmarks\",\"code generation\",\"research\"]","2026-10-01T04:00:00.000Z","2026-10-01T21:33:10.218Z","2026-10-01T21:33:13.357Z","published",null,[],"ai",[26,27,28,29],"ai agents","benchmarks","code generation","research",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2609.39284",0,{"sections":36},[37,41,45,49,54,58,62,67,72,77,82,87,92,97],{"name":38,"slug":24,"count":39,"latest_published_at":40},"AI",5895,"2026-10-02T04:00:00.000Z",{"name":42,"slug":43,"count":44,"latest_published_at":40},"Security","security",836,{"name":46,"slug":47,"count":48,"latest_published_at":40},"Policy","policy",438,{"name":50,"slug":51,"count":52,"latest_published_at":53},"Deals","deals",317,"2026-10-01T22:00:00.000Z",{"name":55,"slug":56,"count":57,"latest_published_at":40},"Hardware","hardware",199,{"name":59,"slug":60,"count":61,"latest_published_at":40},"Science","science",171,{"name":63,"slug":64,"count":65,"latest_published_at":66},"Consumer Tech","consumer-tech",155,"2026-10-01T19:54:10.000Z",{"name":68,"slug":69,"count":70,"latest_published_at":71},"Dev Tools","dev-tools",96,"2026-10-01T16:57:03.000Z",{"name":73,"slug":74,"count":75,"latest_published_at":76},"Software","software",93,"2026-09-30T21:41:11.000Z",{"name":78,"slug":79,"count":80,"latest_published_at":81},"Startups","startups",90,"2026-10-01T21:55:22.000Z",{"name":83,"slug":84,"count":85,"latest_published_at":86},"Gaming","gaming",53,"2026-10-02T02:50:39.000Z",{"name":88,"slug":89,"count":90,"latest_published_at":91},"General","general",50,"2026-09-30T21:37:54.000Z",{"name":93,"slug":94,"count":95,"latest_published_at":96},"Reviews","reviews",31,"2026-09-28T14:31:34.000Z",{"name":98,"slug":99,"count":100,"latest_published_at":101},"How-To","how-to",7,"2026-10-01T09:00:00.000Z"]