[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-scienceflow-sets-new-bar-for-ai-agents-that-run-experiments-solo":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},5034,"scienceflow-sets-new-bar-for-ai-agents-that-run-experiments-solo","ScienceFlow Sets New Bar for AI Agents That Run Experiments Solo","A new autoresearch agent framework tops a leading benchmark by managing long research sessions like recoverable checkpoints instead of one long fragile run.","Researchers have built an AI agent that can run science experiments for a full day without losing its train of thought.\n\nScienceFlow is a framework for LLM agents that conduct autonomous research over long stretches of time, whether that means training machine learning models, running scientific simulations, or solving optimization problems. The core problem it targets: existing autoresearch agents tend to wander into dead ends, forget useful progress, and burn compute without a plan for recovering when things go wrong. ScienceFlow instead breaks research into segments backed by executable, savable states, so the agent can rewind to a good checkpoint or redirect its approach using a mechanism the paper calls Executable-State Transition through Re-Anchoring. A separate controller decides how to spend compute based on what's left in the budget and what's actually been proven to work.\n\nThe headline number is a 70.22 percent Any-Medal score on the full MLE-bench within a 24-hour compute budget, a 4.92 percentage point improvement over the best previously reported result. That's a meaningful jump for a benchmark designed to measure whether an agent can independently produce competitive machine learning solutions, not just plausible-looking code.\n\nThe interesting part isn't the score, it's the architecture choice. Most autoresearch agents treat a research run as one continuous conversation, which means a bad turn early on can poison everything after it. ScienceFlow treats research more like version control: checkpoint, branch, roll back if needed. That's a mundane engineering idea, but it's the kind of unglamorous fix that tends to separate benchmark demos from tools people actually trust with real compute budgets.\n\nWhether this generalizes past curated benchmarks like MLE-bench, where success criteria are already well-defined, remains an open question -- real research rarely comes with a scoreboard telling you when you've found the answer.","[\"ai-agents\",\"machine-learning\",\"autonomous-research\",\"benchmarks\"]","2026-08-17T04:00:00.000Z","2026-08-17T06:39:55.095Z","2026-08-17T06:40:06.993Z","published",null,[],"ai",[26,27,28,29],"ai-agents","machine-learning","autonomous-research","benchmarks",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2608.14354",0,{"sections":36},[37,41,45,50,55,60,65,70,75,80,85,90,95,100],{"name":38,"slug":24,"count":39,"latest_published_at":40},"AI",3293,"2026-08-20T04:00:00.000Z",{"name":42,"slug":43,"count":44,"latest_published_at":40},"Security","security",435,{"name":46,"slug":47,"count":48,"latest_published_at":49},"Policy","policy",210,"2026-08-19T09:32:27.000Z",{"name":51,"slug":52,"count":53,"latest_published_at":54},"Deals","deals",179,"2026-06-29T20:02:07.000Z",{"name":56,"slug":57,"count":58,"latest_published_at":59},"Hardware","hardware",140,"2026-08-19T18:25:42.000Z",{"name":61,"slug":62,"count":63,"latest_published_at":64},"Consumer Tech","consumer-tech",95,"2026-08-18T16:05:00.000Z",{"name":66,"slug":67,"count":68,"latest_published_at":69},"Science","science",90,"2026-08-19T18:41:02.000Z",{"name":71,"slug":72,"count":73,"latest_published_at":74},"Software","software",73,"2026-08-18T07:51:50.000Z",{"name":76,"slug":77,"count":78,"latest_published_at":79},"Dev Tools","dev-tools",69,"2026-08-18T04:00:00.000Z",{"name":81,"slug":82,"count":83,"latest_published_at":84},"Startups","startups",47,"2026-08-19T19:13:46.000Z",{"name":86,"slug":87,"count":88,"latest_published_at":89},"Gaming","gaming",41,"2026-07-09T04:00:00.000Z",{"name":91,"slug":92,"count":93,"latest_published_at":94},"General","general",33,"2026-08-18T22:18:13.000Z",{"name":96,"slug":97,"count":98,"latest_published_at":99},"Reviews","reviews",20,"2026-06-24T12:00:01.000Z",{"name":101,"slug":102,"count":103,"latest_published_at":104},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]