[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-ai-benchmark-puts-models-against-82-unsolved-science-problems":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},10849,"ai-benchmark-puts-models-against-82-unsolved-science-problems","AI Benchmark Puts Models Against 82 Unsolved Science Problems","OpenProblemBench scores AI models on 82 unsolved math and physics problems, and even the best one solves them only 14 percent of the time.","A new benchmark asked AI models to make headway on problems nobody has ever solved, and the best model only got partway there 14 percent of the time.\n\nResearchers built OpenProblemBench from 82 unresolved problems pulled from the math and theoretical physics literature. Each problem includes the background, the assumptions researchers already accept, and however far anyone has gotten, so models aren't starting from zero. Since there's no answer key for a problem nobody has solved, four separate AI models grade each submission on correctness, completeness, and how much real progress it makes. Across seven model configurations tested, GPT-6-Astra posted the highest average judged solve rate at 14.0 percent, full-size open-source models scored 5.5 to 6.7 percent, and smaller Flash-tier models managed only 2.4 to 3.7 percent.\n\nThat gap matters because it separates AI's long-running party trick, reciting known facts, from something closer to research: generating a correct answer where none exists in any training set. The size difference between the top score and the open-model scores also suggests raw scale is still buying real advantage on frontier reasoning problems, not just better test-taking.\n\nStill, judged progress here is decided by other AI models grading each other's homework, not working mathematicians, so a 14 percent solve rate reads less like an AI closing in on an unsolved conjecture and more like a new leaderboard for a notably strange video game.","[\"ai benchmarks\",\"math\",\"theoretical physics\",\"llm evaluation\"]","2026-10-09T04:00:00.000Z","2026-10-09T18:13:44.766Z","2026-10-09T18:13:50.307Z","published",null,[],"ai",[26,27,28,29],"ai benchmarks","math","theoretical physics","llm evaluation",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2610.11118",0,{"sections":36},[37,40,44,49,54,59,63,68,73,78,83,88,93,98],{"name":38,"slug":24,"count":39,"latest_published_at":18},"AI",6603,{"name":41,"slug":42,"count":43,"latest_published_at":18},"Security","security",926,{"name":45,"slug":46,"count":47,"latest_published_at":48},"Policy","policy",486,"2026-10-08T22:40:11.000Z",{"name":50,"slug":51,"count":52,"latest_published_at":53},"Deals","deals",474,"2026-10-08T22:00:00.000Z",{"name":55,"slug":56,"count":57,"latest_published_at":58},"Hardware","hardware",229,"2026-10-08T20:47:10.000Z",{"name":60,"slug":61,"count":62,"latest_published_at":18},"Science","science",192,{"name":64,"slug":65,"count":66,"latest_published_at":67},"Consumer Tech","consumer-tech",181,"2026-10-08T23:26:35.000Z",{"name":69,"slug":70,"count":71,"latest_published_at":72},"Startups","startups",117,"2026-10-08T16:45:00.000Z",{"name":74,"slug":75,"count":76,"latest_published_at":77},"Software","software",114,"2026-10-08T17:57:01.000Z",{"name":79,"slug":80,"count":81,"latest_published_at":82},"Dev Tools","dev-tools",105,"2026-10-07T16:59:11.000Z",{"name":84,"slug":85,"count":86,"latest_published_at":87},"General","general",66,"2026-10-09T04:46:11.000Z",{"name":89,"slug":90,"count":91,"latest_published_at":92},"Gaming","gaming",58,"2026-10-08T20:08:45.000Z",{"name":94,"slug":95,"count":96,"latest_published_at":97},"Reviews","reviews",34,"2026-10-08T14:00:22.000Z",{"name":99,"slug":100,"count":101,"latest_published_at":102},"How-To","how-to",8,"2026-10-05T09:00:00.000Z"]