[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-researchers-catch-vision-ai-making-up-explanations":10,"sections":34},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":29,"feedback":33,"feedback_at":22,"cost_usd":33,"total_tokens":33},6589,"researchers-catch-vision-ai-making-up-explanations","Researchers Catch Vision AI Making Up Explanations","EDCT-Bench edits images and checks whether a model's stated reasoning actually tracks the change, and most vision-language models fail.","Ask a vision-language model why it gave an answer, and it will happily explain itself. A new benchmark shows that explanation is often disconnected from what's actually in the image.\n\nResearchers built EDCT-Bench using a method they call Explanation-Driven Counterfactual Testing. It pulls out the visual details a model cites in its explanation, makes a small verified edit to that part of the image, and checks whether the model's new answer and explanation actually track the change. The benchmark spans three domains: OK-VQA for general visual question answering, DriveLM for driving scenarios, and 3DSRBench for 3D spatial reasoning. Across the models tested, answers and explanations frequently stayed the same even after the underlying visual evidence changed.\n\nThat gap matters most in DriveLM's territory. A driving system that explains a decision by citing a pedestrian or a stop sign needs that explanation tied to what the camera actually sees, not a plausible story bolted on afterward. The researchers' fine-tuning experiments suggest these counterfactual edits also work as useful training data, not just a diagnostic tool.\n\nIt's the same trust problem language models have with hallucinated citations, translated to pixels: a fluent explanation is not evidence the model looked at the right thing.","[\"ai\",\"vision-language-models\",\"benchmarks\",\"ai-safety\"]","2026-09-17T04:00:00.000Z","2026-09-18T01:30:11.722Z","2026-09-18T01:30:23.671Z","published",null,[],"ai",[24,26,27,28],"vision-language-models","benchmarks","ai-safety",[30],{"name":31,"url":32},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2609.17953",0,{"sections":35},[36,40,44,49,54,58,62,67,72,76,81,86,91,96],{"name":37,"slug":24,"count":38,"latest_published_at":39},"AI",3853,"2026-09-17T08:27:09.000Z",{"name":41,"slug":42,"count":43,"latest_published_at":18},"Security","security",648,{"name":45,"slug":46,"count":47,"latest_published_at":48},"Policy","policy",338,"2026-09-11T04:00:00.000Z",{"name":50,"slug":51,"count":52,"latest_published_at":53},"Deals","deals",179,"2026-06-29T20:02:07.000Z",{"name":55,"slug":56,"count":57,"latest_published_at":18},"Hardware","hardware",154,{"name":59,"slug":60,"count":61,"latest_published_at":18},"Science","science",114,{"name":63,"slug":64,"count":65,"latest_published_at":66},"Consumer Tech","consumer-tech",99,"2026-09-09T17:27:33.000Z",{"name":68,"slug":69,"count":70,"latest_published_at":71},"Software","software",75,"2026-09-10T20:41:21.000Z",{"name":73,"slug":74,"count":75,"latest_published_at":18},"Dev Tools","dev-tools",73,{"name":77,"slug":78,"count":79,"latest_published_at":80},"Startups","startups",55,"2026-09-09T23:14:29.000Z",{"name":82,"slug":83,"count":84,"latest_published_at":85},"Gaming","gaming",43,"2026-09-10T12:18:06.000Z",{"name":87,"slug":88,"count":89,"latest_published_at":90},"General","general",41,"2026-09-08T01:57:23.000Z",{"name":92,"slug":93,"count":94,"latest_published_at":95},"Reviews","reviews",20,"2026-06-24T12:00:01.000Z",{"name":97,"slug":98,"count":99,"latest_published_at":100},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]