[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-selfsearch-lets-ai-coding-agents-improve-without-reward-signals":10,"sections":40},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":30,"tags":31,"sources":35,"feedback":39,"feedback_at":22,"cost_usd":39,"total_tokens":39},8536,"selfsearch-lets-ai-coding-agents-improve-without-reward-signals","SelfSearch Lets AI Coding Agents Improve Without Reward Signals","A new technique called SelfSearch lets AI coding agents improve by learning from their own past attempts, without expensive reward-based search.","Researchers have built AI coding agents that get better at their jobs by studying their own trial-and-error, no scorekeeping required.\n\nA new method called SelfSearch lets AI agents rewrite their own instructions and tools by reviewing logs of earlier self-modification attempts, instead of being graded against a benchmark during the search itself. Tested across six combinations of models and benchmarks, the approach improved average task success in every case, with the best-performing agent gaining 11.2 percentage points on Terminal-Bench 2.1, a benchmark that scores how well an AI agent completes real command-line and software tasks. On SWE-bench Multilingual, which tests an agent's ability to fix real bugs in codebases written in multiple programming languages, one agent improved success by 5.0 percentage points while cutting execution costs by 38.5% on tasks it could already solve. The whole search process, which produced a harness matching the top-scoring system in a public nine-harness comparison, cost just $4.03.\n\nMost agent self-improvement work still leans on repeatedly running and grading agents against the very tasks they're being optimized for, which is expensive and risks overfitting. SelfSearch skips that grading step, using only records of what the agent tried and what happened, which suggests agents can get both better and cheaper without a benchmark babysitting every step of the search.\n\nThat's a notable efficiency claim: a few dollars in search cost landing results on par with harnesses that typically burn far more compute to tune.","[\"ai\",\"coding-agents\",\"benchmarks\",\"ai-research\"]","2026-09-30T04:00:00.000Z","2026-09-30T09:47:13.536Z","2026-09-30T09:47:15.796Z","published",null,[24],{"id":25,"reviewer":26,"round":27,"reason":28,"status":29},"editor-r1","editor",1,"Add a brief explanation of what Terminal-Bench 2.1 and SWE-bench Multilingual actually test (e.g., command-line\u002Fcoding task completion) so the percentage-point and cost figures are meaningful to readers unfamiliar with these benchmarks.","resolved","ai",[30,32,33,34],"coding-agents","benchmarks","ai-research",[36],{"name":37,"url":38},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2609.37968",0,{"sections":41},[42,45,49,53,58,63,68,73,78,82,87,92,97,102],{"name":43,"slug":30,"count":44,"latest_published_at":18},"AI",5104,{"name":46,"slug":47,"count":48,"latest_published_at":18},"Security","security",785,{"name":50,"slug":51,"count":52,"latest_published_at":18},"Policy","policy",417,{"name":54,"slug":55,"count":56,"latest_published_at":57},"Deals","deals",284,"2026-09-29T21:00:00.000Z",{"name":59,"slug":60,"count":61,"latest_published_at":62},"Hardware","hardware",194,"2026-09-29T13:16:04.000Z",{"name":64,"slug":65,"count":66,"latest_published_at":67},"Science","science",154,"2026-09-28T13:19:18.000Z",{"name":69,"slug":70,"count":71,"latest_published_at":72},"Consumer Tech","consumer-tech",142,"2026-09-29T18:38:03.000Z",{"name":74,"slug":75,"count":76,"latest_published_at":77},"Software","software",91,"2026-09-25T20:55:00.000Z",{"name":79,"slug":80,"count":81,"latest_published_at":18},"Dev Tools","dev-tools",90,{"name":83,"slug":84,"count":85,"latest_published_at":86},"Startups","startups",83,"2026-09-29T21:51:36.000Z",{"name":88,"slug":89,"count":90,"latest_published_at":91},"General","general",49,"2026-09-28T16:44:57.000Z",{"name":93,"slug":94,"count":95,"latest_published_at":96},"Gaming","gaming",48,"2026-09-25T18:35:21.000Z",{"name":98,"slug":99,"count":100,"latest_published_at":101},"Reviews","reviews",31,"2026-09-28T14:31:34.000Z",{"name":103,"slug":104,"count":105,"latest_published_at":106},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]