[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-researchers-find-exactly-where-multimodal-ai-models-break":10,"sections":34},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":29,"feedback":33,"feedback_at":22,"cost_usd":33,"total_tokens":33},9073,"researchers-find-exactly-where-multimodal-ai-models-break","Researchers Find Exactly Where Multimodal AI Models Break","A new benchmark called CADET shows many AI reasoning failures actually stem from bad upstream inputs, not weak logic itself.","A new benchmark diagnoses whether AI models fail at reasoning itself or just inherit someone else's mistake.\n\nResearchers built a diagnostic framework that pulls apart compositional tasks - the multi-step problems where a model has to see something, locate it, track it over time, then reason about it - into their prerequisite steps. Instead of just scoring whether a multimodal AI model (MLLM) gets the final answer right, the framework feeds each model correct, incorrect, or no prerequisite information and measures how the answer changes. That diagnostic method is instantiated in CADET, a benchmark of 10 composite tasks broken into 46 unit tasks, covering perception, spatial, temporal, and cognitive skills, built from over 33,000 human-annotated questions. Testing frontier MLLMs with it found that handing a model correct prerequisites wiped out 54 percent of its errors on cognitive tasks alone, enough to flip cognitive reasoning from the weakest category to stronger than spatial and temporal ones.\n\nThat's a big deal for how we think about model failure. A low score on a reasoning benchmark might mean a model can't reason at all - or it might mean the model reasoned fine once it had the right inputs, and the real bug lives upstream in perception or retrieval. The study also found the blame is concentrated: giving a model just the single most important prerequisite recovers 84 percent of the benefit of giving it everything, meaning most failures trace back to one or two weak links, not a uniformly bad pipeline.\n\nEnd-to-end accuracy scores have told us models fail for years; this is one of the more rigorous attempts to say exactly where and why - though whether frontier labs will bother fixing the specific links it identifies is a separate question.","[\"ai\",\"benchmarks\",\"research\",\"multimodal-models\"]","2026-10-01T04:00:00.000Z","2026-10-01T18:31:52.548Z","2026-10-01T18:31:58.730Z","published",null,[],"ai",[24,26,27,28],"benchmarks","research","multimodal-models",[30],{"name":31,"url":32},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2609.38851",0,{"sections":35},[36,39,43,48,53,58,62,67,72,76,81,86,91,96],{"name":37,"slug":24,"count":38,"latest_published_at":18},"AI",5488,{"name":40,"slug":41,"count":42,"latest_published_at":18},"Security","security",809,{"name":44,"slug":45,"count":46,"latest_published_at":47},"Policy","policy",429,"2026-10-01T02:26:17.000Z",{"name":49,"slug":50,"count":51,"latest_published_at":52},"Deals","deals",298,"2026-09-30T21:00:26.000Z",{"name":54,"slug":55,"count":56,"latest_published_at":57},"Hardware","hardware",196,"2026-09-30T13:00:00.000Z",{"name":59,"slug":60,"count":61,"latest_published_at":18},"Science","science",162,{"name":63,"slug":64,"count":65,"latest_published_at":66},"Consumer Tech","consumer-tech",149,"2026-09-30T22:57:11.000Z",{"name":68,"slug":69,"count":70,"latest_published_at":71},"Dev Tools","dev-tools",93,"2026-10-01T02:30:48.000Z",{"name":73,"slug":74,"count":70,"latest_published_at":75},"Software","software","2026-09-30T21:41:11.000Z",{"name":77,"slug":78,"count":79,"latest_published_at":80},"Startups","startups",84,"2026-09-30T20:39:09.000Z",{"name":82,"slug":83,"count":84,"latest_published_at":85},"Gaming","gaming",51,"2026-09-30T16:24:30.000Z",{"name":87,"slug":88,"count":89,"latest_published_at":90},"General","general",50,"2026-09-30T21:37:54.000Z",{"name":92,"slug":93,"count":94,"latest_published_at":95},"Reviews","reviews",31,"2026-09-28T14:31:34.000Z",{"name":97,"slug":98,"count":99,"latest_published_at":100},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]