[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-researchers-find-failed-ai-prompts-are-worth-keeping":10,"sections":34},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":29,"feedback":33,"feedback_at":22,"cost_usd":33,"total_tokens":33},8758,"researchers-find-failed-ai-prompts-are-worth-keeping","Researchers Find Failed AI Prompts Are Worth Keeping","Mara Chain reuses rejected AI prompt and harness edits instead of discarding them, cutting rollouts needed to improve performance.","A new paper argues that AI teams tuning prompts and agent harnesses have been throwing away their most useful data: the failed attempts.\n\nResearchers introduce Mara Chain, a refinement procedure for optimizing deployed AI systems, meaning the prompts, skills, harnesses, and code that increasingly substitute for retraining model weights. Standard propose-evaluate-select loops score candidate edits and discard anything that doesn't meet an acceptance bar, but Mara Chain instead keeps rejected candidates and iteratively refines them using evidence from earlier attempts, capping each refinement chain at a fixed depth and trimming the pool with Pareto-filtered Top-N selection. Tested on AppWorld skill optimization, TerminalBench 2.1 harness optimization, and MuSiQue retrieval-pipeline optimization, it beat existing optimizers GEPA, ACE, and SkillOpt-Lite by up to 20.5% in relative performance on AppWorld, reaching the target score with 65.5% fewer rollouts than GEPA. On TerminalBench 2.1 it raised pass rates by 20.2 and 22.5 percentage points over two rival methods, and on MuSiQue it beat a hand-written retrieval pipeline by 0.104 on nDCG@10 and 0.131 on Recall@10.\n\nThe interesting claim isn't the benchmark wins, it is the premise behind them: a rejected configuration still carries information about which failure modes to avoid, and tossing it forces the next round of proposals to rediscover the same dead ends. For teams running automated prompt or harness tuning at scale, that is a direct cut in wasted rollouts, and rollouts cost real compute. It also reflects a wider shift toward treating prompt and harness design as a formal search problem rather than manual trial and error.\n\nThe gains come from benchmarks chosen by the paper's own authors, so the harder test, as always, is whether other teams see the same numbers when they try it on their own systems.","[\"ai\",\"ai-agents\",\"prompt-engineering\",\"research\"]","2026-09-30T04:00:00.000Z","2026-10-01T00:15:26.099Z","2026-10-01T00:15:31.900Z","published",null,[],"ai",[24,26,27,28],"ai-agents","prompt-engineering","research",[30],{"name":31,"url":32},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2609.35855",0,{"sections":35},[36,40,45,50,55,59,63,67,72,76,81,86,91,96],{"name":37,"slug":24,"count":38,"latest_published_at":39},"AI",5214,"2026-09-30T13:00:00.000Z",{"name":41,"slug":42,"count":43,"latest_published_at":44},"Security","security",793,"2026-09-30T12:55:00.000Z",{"name":46,"slug":47,"count":48,"latest_published_at":49},"Policy","policy",419,"2026-09-30T12:24:32.000Z",{"name":51,"slug":52,"count":53,"latest_published_at":54},"Deals","deals",292,"2026-09-30T14:15:18.000Z",{"name":56,"slug":57,"count":58,"latest_published_at":39},"Hardware","hardware",196,{"name":60,"slug":61,"count":62,"latest_published_at":18},"Science","science",155,{"name":64,"slug":65,"count":66,"latest_published_at":39},"Consumer Tech","consumer-tech",144,{"name":68,"slug":69,"count":70,"latest_published_at":71},"Dev Tools","dev-tools",91,"2026-09-30T12:58:00.000Z",{"name":73,"slug":74,"count":70,"latest_published_at":75},"Software","software","2026-09-25T20:55:00.000Z",{"name":77,"slug":78,"count":79,"latest_published_at":80},"Startups","startups",83,"2026-09-29T21:51:36.000Z",{"name":82,"slug":83,"count":84,"latest_published_at":85},"General","general",49,"2026-09-28T16:44:57.000Z",{"name":87,"slug":88,"count":89,"latest_published_at":90},"Gaming","gaming",48,"2026-09-25T18:35:21.000Z",{"name":92,"slug":93,"count":94,"latest_published_at":95},"Reviews","reviews",31,"2026-09-28T14:31:34.000Z",{"name":97,"slug":98,"count":99,"latest_published_at":100},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]