[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-ai-coding-agent-writes-its-own-tests-first":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},5483,"ai-coding-agent-writes-its-own-tests-first","AI Coding Agent Writes Its Own Tests First","TDD-Agent has the model write tests before code, then refines both together with execution feedback, beating retrieval and agent baselines on repo-level tasks.","A new AI coding agent writes its own tests before it writes a single line of the code those tests are meant to check.\n\nResearchers call it TDD-Agent, and it borrows the test-driven development habit familiar to any working programmer: define what \"correct\" looks like, then build toward it. The model first generates executable tests to pin down expected behavior, then alternates between refining its code and its tests using real execution feedback, rather than treating tests as a one-time check run after the fact. On the LiveCodeBench benchmark, a simpler prompt-only version of this approach already beat standard reasoning-based prompting. The full agent, tested on the repository-level benchmark RepoEval, outperformed both retrieval-based and existing agent-based baselines.\n\nThis matters because most AI coding tools still use tests as an afterthought - generated once, then checked and discarded. If a model's first-draft test is wrong or incomplete, that flaw quietly poisons every downstream check. TDD-Agent's iterative loop instead lets tests improve alongside the code, and the researchers report that pass rates, coverage, and mutation scores all rise as a result - tests functioning as reasoning tools, not just gatekeepers.\n\nIt is a modest, sensible idea dressed up in agent-framework language: make the AI decide what \"done\" means before it starts, the way a disciplined engineer would. Whether it holds up outside curated benchmarks, on the messy, half-documented repos most developers actually work in, is the harder question this paper doesn't answer.","[\"ai\",\"code-generation\",\"software-testing\",\"llm-agents\"]","2026-08-18T04:00:00.000Z","2026-08-18T22:07:54.793Z","2026-08-18T22:08:06.753Z","published",null,[],"dev-tools",[26,27,28,29],"ai","code-generation","software-testing","llm-agents",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2608.16742",0,{"sections":36},[37,41,45,50,55,60,65,70,75,78,83,88,93,98],{"name":38,"slug":26,"count":39,"latest_published_at":40},"AI",3293,"2026-08-20T04:00:00.000Z",{"name":42,"slug":43,"count":44,"latest_published_at":40},"Security","security",435,{"name":46,"slug":47,"count":48,"latest_published_at":49},"Policy","policy",210,"2026-08-19T09:32:27.000Z",{"name":51,"slug":52,"count":53,"latest_published_at":54},"Deals","deals",179,"2026-06-29T20:02:07.000Z",{"name":56,"slug":57,"count":58,"latest_published_at":59},"Hardware","hardware",140,"2026-08-19T18:25:42.000Z",{"name":61,"slug":62,"count":63,"latest_published_at":64},"Consumer Tech","consumer-tech",95,"2026-08-18T16:05:00.000Z",{"name":66,"slug":67,"count":68,"latest_published_at":69},"Science","science",90,"2026-08-19T18:41:02.000Z",{"name":71,"slug":72,"count":73,"latest_published_at":74},"Software","software",73,"2026-08-18T07:51:50.000Z",{"name":76,"slug":24,"count":77,"latest_published_at":18},"Dev Tools",69,{"name":79,"slug":80,"count":81,"latest_published_at":82},"Startups","startups",47,"2026-08-19T19:13:46.000Z",{"name":84,"slug":85,"count":86,"latest_published_at":87},"Gaming","gaming",41,"2026-07-09T04:00:00.000Z",{"name":89,"slug":90,"count":91,"latest_published_at":92},"General","general",33,"2026-08-18T22:18:13.000Z",{"name":94,"slug":95,"count":96,"latest_published_at":97},"Reviews","reviews",20,"2026-06-24T12:00:01.000Z",{"name":99,"slug":100,"count":101,"latest_published_at":102},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]