[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-why-ai-vision-benchmarks-may-be-measuring-text-not-sight":10,"sections":34},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":29,"feedback":33,"feedback_at":22,"cost_usd":33,"total_tokens":33},9294,"why-ai-vision-benchmarks-may-be-measuring-text-not-sight","Why AI Vision Benchmarks May Be Measuring Text, Not Sight","A new benchmark finds top AI vision models lose over half their accuracy once polar layouts remove the text-coordinate shortcut they rely on.","Leading multimodal AI models ace grid-based visual reasoning tests by quietly converting pictures into coordinate lists instead of actually looking at them.\n\nResearchers built Polaris-Bench, a set of 53 visual reasoning tasks reformulated in polar coordinates, each paired with a Cartesian version that tests identical logic. Polar layouts break the neat rows-and-columns structure models exploit to translate an image into text coordinates. Across 14 state-of-the-art multimodal models, accuracy on standard Cartesian grids ranged from 69% to 83%. On the Polar versions of the same tasks, scores collapsed to 31-39%, and prompting tricks meant to boost reasoning barely moved the needle.\n\nThe gap suggests a chunk of what benchmarks call visual reasoning has really been text-based deduction wearing a picture's clothing. Human testers held 88.8% accuracy on the same polar tasks, which rules out the simple explanation that this is just a hard visual problem and points squarely at a shortcut baked into how these models process grids.\n\nIt's a useful gut check for anyone citing benchmark leaderboards as proof of visual intelligence: a lot of that score may belong to a model's grasp of coordinates, not its eyes.","[\"ai\",\"benchmarks\",\"computer-vision\",\"multimodal-models\"]","2026-10-01T04:00:00.000Z","2026-10-02T07:27:20.820Z","2026-10-02T07:27:26.101Z","published",null,[],"ai",[24,26,27,28],"benchmarks","computer-vision","multimodal-models",[30],{"name":31,"url":32},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2605.09883",0,{"sections":35},[36,39,43,47,52,57,61,66,71,75,80,85,90,95],{"name":37,"slug":24,"count":38,"latest_published_at":18},"AI",5671,{"name":40,"slug":41,"count":42,"latest_published_at":18},"Security","security",820,{"name":44,"slug":45,"count":46,"latest_published_at":18},"Policy","policy",430,{"name":48,"slug":49,"count":50,"latest_published_at":51},"Deals","deals",298,"2026-09-30T21:00:26.000Z",{"name":53,"slug":54,"count":55,"latest_published_at":56},"Hardware","hardware",196,"2026-09-30T13:00:00.000Z",{"name":58,"slug":59,"count":60,"latest_published_at":18},"Science","science",165,{"name":62,"slug":63,"count":64,"latest_published_at":65},"Consumer Tech","consumer-tech",149,"2026-09-30T22:57:11.000Z",{"name":67,"slug":68,"count":69,"latest_published_at":70},"Dev Tools","dev-tools",93,"2026-10-01T02:30:48.000Z",{"name":72,"slug":73,"count":69,"latest_published_at":74},"Software","software","2026-09-30T21:41:11.000Z",{"name":76,"slug":77,"count":78,"latest_published_at":79},"Startups","startups",84,"2026-09-30T20:39:09.000Z",{"name":81,"slug":82,"count":83,"latest_published_at":84},"Gaming","gaming",51,"2026-09-30T16:24:30.000Z",{"name":86,"slug":87,"count":88,"latest_published_at":89},"General","general",50,"2026-09-30T21:37:54.000Z",{"name":91,"slug":92,"count":93,"latest_published_at":94},"Reviews","reviews",31,"2026-09-28T14:31:34.000Z",{"name":96,"slug":97,"count":98,"latest_published_at":99},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]