[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-new-benchmark-shows-ai-models-fail-basic-visual-logic-tests":10,"sections":34},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":29,"feedback":33,"feedback_at":22,"cost_usd":33,"total_tokens":33},9745,"new-benchmark-shows-ai-models-fail-basic-visual-logic-tests","New Benchmark Shows AI Models Fail Basic Visual Logic Tests","Hob-VL tests whether AI can combine simple visual facts with Boolean logic, and most models score barely better than a coin flip.","AI models that can recognize a cat in a photo still struggle to reason logically about what they are seeing.\n\nA new benchmark called Hob-VL tests whether vision-language models can combine simple visual observations using Boolean logic - AND, OR, NOT, and nested combinations - rather than just spotting objects. The dataset includes 6,000 human-verified Yes\u002FNo questions built from combinations of ten visual statements, spanning 1,000 generated scenes and 46 labeled photographs, plus 1,000 questions asking models to identify the one object matching a given description. Questions are deliberately built with misleading local cues and nested logical operations, presented in both symbolic and plain-language form. Across eight model setups with little or no step-by-step reasoning enabled, Boolean-question accuracy landed between 48.52% and 50.57% - essentially a coin flip on a yes\u002Fno task - while object-identification accuracy topped out at 43.0%. Even a reasoning-enabled GLM configuration only improved unevenly, with researchers reporting persistent errors and inconsistent answers to logically equivalent questions.\n\nThat gap matters because compositional reasoning, not just object recognition, is what separates a model that can describe a photo from one that can be trusted to act on what it sees - in search, robotics, or any tool that chains visual facts into a decision. Near-random scores on a benchmark built from simple logical combinations suggest current vision-language models are still pattern-matching their way through images rather than reasoning about them.\n\nA model that answers 'is there a red car or a blue bike' correctly but flips to wrong on the logically identical 'is it false that there is neither a red car nor a blue bike' was never really reasoning - it was guessing with good vocabulary.","[\"ai\",\"benchmarks\",\"computer-vision\",\"multimodal-ai\"]","2026-10-02T04:00:00.000Z","2026-10-03T08:47:27.496Z","2026-10-03T08:47:27.678Z","published",null,[],"ai",[24,26,27,28],"benchmarks","computer-vision","multimodal-ai",[30],{"name":31,"url":32},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2610.01605",0,{"sections":35},[36,39,43,47,52,56,60,65,70,75,80,85,90,95],{"name":37,"slug":24,"count":38,"latest_published_at":18},"AI",6041,{"name":40,"slug":41,"count":42,"latest_published_at":18},"Security","security",848,{"name":44,"slug":45,"count":46,"latest_published_at":18},"Policy","policy",439,{"name":48,"slug":49,"count":50,"latest_published_at":51},"Deals","deals",317,"2026-10-01T22:00:00.000Z",{"name":53,"slug":54,"count":55,"latest_published_at":18},"Hardware","hardware",199,{"name":57,"slug":58,"count":59,"latest_published_at":18},"Science","science",176,{"name":61,"slug":62,"count":63,"latest_published_at":64},"Consumer Tech","consumer-tech",155,"2026-10-01T19:54:10.000Z",{"name":66,"slug":67,"count":68,"latest_published_at":69},"Dev Tools","dev-tools",96,"2026-10-01T16:57:03.000Z",{"name":71,"slug":72,"count":73,"latest_published_at":74},"Software","software",93,"2026-09-30T21:41:11.000Z",{"name":76,"slug":77,"count":78,"latest_published_at":79},"Startups","startups",90,"2026-10-01T21:55:22.000Z",{"name":81,"slug":82,"count":83,"latest_published_at":84},"Gaming","gaming",53,"2026-10-02T02:50:39.000Z",{"name":86,"slug":87,"count":88,"latest_published_at":89},"General","general",50,"2026-09-30T21:37:54.000Z",{"name":91,"slug":92,"count":93,"latest_published_at":94},"Reviews","reviews",31,"2026-09-28T14:31:34.000Z",{"name":96,"slug":97,"count":98,"latest_published_at":99},"How-To","how-to",7,"2026-10-01T09:00:00.000Z"]