[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-study-finds-ai-explanations-dont-improve-experiment-forecasts":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},9457,"study-finds-ai-explanations-dont-improve-experiment-forecasts","Study Finds AI Explanations Don't Improve Experiment Forecasts","New research finds AI agents' experiment explanations add little predictive value beyond a plain description of what's being tested.","A new benchmark finds that AI research agents' explanations for their own experiments mostly fail to make outcomes any easier to predict.\n\nResearchers built a protocol that compares paired forecasts of the same experiment, made by the same forecaster, where only the input changes: a plain description, a matched explanation, or added 'donor' context pulled from elsewhere. They tested it across 336 scenarios in a controlled learning setup, 12 toxicity-prediction tasks from the Tox21 dataset, and 24 machine learning benchmarks from OpenML, checking five things including whether an agent's explanation measurably improved accuracy over the bare description. The core preregistered test of whether explanations beat descriptions came back unmet on both Tox21 and OpenML. Results also depended heavily on which model wrote the explanation: DeepSeek V4 Pro's explanations cut a secondary drift metric on Tox21 by more than half, while a DeepSeek V4 Flash version made point predictions worse, pushing error from .01823 to .02020.\n\nThat matters because AI labs are increasingly marketing research agents on the idea that they don't just spit out results, they reason through them in language a person can check. This study suggests that reasoning-sounding text and predictively useful text are not the same thing, and current models' natural-language explanations did not reliably clear that bar at the resolutions tested. The one unambiguous win came from an explanation a human researcher wrote by hand, which cut prediction error by 2.6 percentage points versus a plain description.\n\nIf a person's one-paragraph explanation still beats the model's, the model isn't explaining its work so much as describing it with extra confidence.","[\"ai-agents\",\"machine-learning\",\"benchmarks\",\"research\"]","2026-10-02T04:00:00.000Z","2026-10-02T19:59:34.081Z","2026-10-02T19:59:40.439Z","published",null,[],"ai",[26,27,28,29],"ai-agents","machine-learning","benchmarks","research",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2610.00314",0,{"sections":36},[37,40,44,49,54,59,64,69,74,79,84,89,94,99],{"name":38,"slug":24,"count":39,"latest_published_at":18},"AI",5764,{"name":41,"slug":42,"count":43,"latest_published_at":18},"Security","security",831,{"name":45,"slug":46,"count":47,"latest_published_at":48},"Policy","policy",437,"2026-10-01T18:10:00.000Z",{"name":50,"slug":51,"count":52,"latest_published_at":53},"Deals","deals",317,"2026-10-01T22:00:00.000Z",{"name":55,"slug":56,"count":57,"latest_published_at":58},"Hardware","hardware",198,"2026-10-01T17:38:48.000Z",{"name":60,"slug":61,"count":62,"latest_published_at":63},"Science","science",168,"2026-10-01T18:35:55.000Z",{"name":65,"slug":66,"count":67,"latest_published_at":68},"Consumer Tech","consumer-tech",155,"2026-10-01T19:54:10.000Z",{"name":70,"slug":71,"count":72,"latest_published_at":73},"Dev Tools","dev-tools",96,"2026-10-01T16:57:03.000Z",{"name":75,"slug":76,"count":77,"latest_published_at":78},"Software","software",93,"2026-09-30T21:41:11.000Z",{"name":80,"slug":81,"count":82,"latest_published_at":83},"Startups","startups",90,"2026-10-01T21:55:22.000Z",{"name":85,"slug":86,"count":87,"latest_published_at":88},"Gaming","gaming",53,"2026-10-02T02:50:39.000Z",{"name":90,"slug":91,"count":92,"latest_published_at":93},"General","general",50,"2026-09-30T21:37:54.000Z",{"name":95,"slug":96,"count":97,"latest_published_at":98},"Reviews","reviews",31,"2026-09-28T14:31:34.000Z",{"name":100,"slug":101,"count":102,"latest_published_at":103},"How-To","how-to",7,"2026-10-01T09:00:00.000Z"]