[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-benchmark-finds-ai-agents-fumble-scientific-software-tasks":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},8962,"benchmark-finds-ai-agents-fumble-scientific-software-tasks","Benchmark Finds AI Agents Fumble Scientific Software Tasks","A new 146-task benchmark shows even top AI models struggle to complete real scientific workflows like molecular drawing and statistical analysis.","A new benchmark says today's AI agents still can't reliably run a lab's software stack.\n\nResearchers built OSWorld-Science, a benchmark of 146 tasks testing whether agents built on vision-language models can actually operate scientific software, not just talk about it. The tasks span molecular drawing and retrosynthesis, pathology image analysis, statistical computing, and physical simulation, and were developed with input from domain experts rather than pulled off the shelf. Instead of grading on vibes, the evaluators inspect the actual artifacts an agent produces, such as a molecular structure, a segmentation mask, a plot, or a numerical result, and hand out partial credit for near-misses. The team ran 12 vision-language models through a purpose-built harness that logs every step of the agent's interaction loop.\n\nThe finding that matters: even well-equipped, state-of-the-art models still struggle with tasks a competent lab tech would treat as routine. That's worth noting as software vendors rush to bolt agent features onto scientific tools, since it suggests the gap between a slick chat demo and dependably running an experiment is still wide. The researchers also tie performance to factors like reasoning effort and context length, which matters for anyone deciding how much compute these agents are worth.\n\nFile it next to every other computer-use benchmark that promised agents could handle any app: the demo is always smoother than the real interface.","[\"ai-agents\",\"benchmarks\",\"scientific-computing\",\"computer-use-agents\"]","2026-10-01T04:00:00.000Z","2026-10-01T13:00:06.863Z","2026-10-01T13:00:10.803Z","published",null,[],"ai",[26,27,28,29],"ai-agents","benchmarks","scientific-computing","computer-use-agents",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2609.39903",0,{"sections":36},[37,40,44,49,54,59,63,68,73,77,82,87,92,97],{"name":38,"slug":24,"count":39,"latest_published_at":18},"AI",5455,{"name":41,"slug":42,"count":43,"latest_published_at":18},"Security","security",805,{"name":45,"slug":46,"count":47,"latest_published_at":48},"Policy","policy",429,"2026-10-01T02:26:17.000Z",{"name":50,"slug":51,"count":52,"latest_published_at":53},"Deals","deals",298,"2026-09-30T21:00:26.000Z",{"name":55,"slug":56,"count":57,"latest_published_at":58},"Hardware","hardware",196,"2026-09-30T13:00:00.000Z",{"name":60,"slug":61,"count":62,"latest_published_at":18},"Science","science",159,{"name":64,"slug":65,"count":66,"latest_published_at":67},"Consumer Tech","consumer-tech",149,"2026-09-30T22:57:11.000Z",{"name":69,"slug":70,"count":71,"latest_published_at":72},"Dev Tools","dev-tools",93,"2026-10-01T02:30:48.000Z",{"name":74,"slug":75,"count":71,"latest_published_at":76},"Software","software","2026-09-30T21:41:11.000Z",{"name":78,"slug":79,"count":80,"latest_published_at":81},"Startups","startups",84,"2026-09-30T20:39:09.000Z",{"name":83,"slug":84,"count":85,"latest_published_at":86},"Gaming","gaming",51,"2026-09-30T16:24:30.000Z",{"name":88,"slug":89,"count":90,"latest_published_at":91},"General","general",50,"2026-09-30T21:37:54.000Z",{"name":93,"slug":94,"count":95,"latest_published_at":96},"Reviews","reviews",31,"2026-09-28T14:31:34.000Z",{"name":98,"slug":99,"count":100,"latest_published_at":101},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]