[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-chain-of-thought-prompts-can-break-vlm-answer-scoring":10,"sections":34},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":29,"feedback":33,"feedback_at":22,"cost_usd":33,"total_tokens":33},7866,"chain-of-thought-prompts-can-break-vlm-answer-scoring","Chain-of-thought Prompts Can Break VLM Answer Scoring","Grading vision-language models on reasoning-cue logits instead of finished answers crashed ScienceQA accuracy from 81 percent to 45 percent.","A widely used way of grading multiple-choice AI answers has a hidden bug that can make a working model look like it is guessing at random.\n\nResearchers tested Qwen2.5-VL-7B on ScienceQA using a common evaluation shortcut: a scorer tells the model to reason step by step, then reads the answer straight from the model's output before any reasoning actually happens. Accuracy collapsed from 80.76 percent to 45.48 percent. Across five reshuffled versions of the same questions, 93.54 percent of the graded answers landed on whatever option sat in the first slot, regardless of its content. That is not the model failing to know the answer. Linear probes run on the same hidden states recovered 78.94 percent accuracy, and letting the model actually finish its reasoning before grading restored 75.24 percent.\n\nThe gap traces to where probability mass goes. Right after a reasoning cue, the model's attention is aimed at generating the next word of an explanation, not committing to an answer token. The correct answer is still sitting in the network's later layers. The scorer is just not reading it.\n\nThis matters because plenty of published vision-language-model benchmark numbers likely use this exact shortcut: append a reasoning instruction, then grab logits without letting the model reason. If so, some of the capability gaps reported between models may be measuring evaluation plumbing rather than actual knowledge.\n\nThe effect is not universal, and it varies by dataset and model. But given how many labs quietly reuse the same evaluation scripts, this is the kind of methodological crack worth checking before trusting the next leaderboard.","[\"ai\",\"vision-language-models\",\"benchmarks\",\"evaluation\"]","2026-09-25T04:00:00.000Z","2026-09-26T03:38:09.966Z","2026-09-26T03:38:14.305Z","published",null,[],"ai",[24,26,27,28],"vision-language-models","benchmarks","evaluation",[30],{"name":31,"url":32},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2609.29278",0,{"sections":35},[36,40,45,50,55,60,65,70,75,80,85,90,95,100],{"name":37,"slug":24,"count":38,"latest_published_at":39},"AI",4575,"2026-09-25T20:35:15.000Z",{"name":41,"slug":42,"count":43,"latest_published_at":44},"Security","security",741,"2026-09-25T15:52:13.000Z",{"name":46,"slug":47,"count":48,"latest_published_at":49},"Policy","policy",392,"2026-09-25T18:44:30.000Z",{"name":51,"slug":52,"count":53,"latest_published_at":54},"Deals","deals",256,"2026-09-25T17:00:53.000Z",{"name":56,"slug":57,"count":58,"latest_published_at":59},"Hardware","hardware",185,"2026-09-25T15:00:22.000Z",{"name":61,"slug":62,"count":63,"latest_published_at":64},"Science","science",142,"2026-09-25T14:07:46.000Z",{"name":66,"slug":67,"count":68,"latest_published_at":69},"Consumer Tech","consumer-tech",132,"2026-09-25T15:30:00.000Z",{"name":71,"slug":72,"count":73,"latest_published_at":74},"Software","software",89,"2026-09-25T19:07:10.000Z",{"name":76,"slug":77,"count":78,"latest_published_at":79},"Dev Tools","dev-tools",82,"2026-09-25T09:59:40.000Z",{"name":81,"slug":82,"count":83,"latest_published_at":84},"Startups","startups",76,"2026-09-25T18:33:59.000Z",{"name":86,"slug":87,"count":88,"latest_published_at":89},"Gaming","gaming",48,"2026-09-25T18:35:21.000Z",{"name":91,"slug":92,"count":93,"latest_published_at":94},"General","general",46,"2026-09-25T02:12:57.000Z",{"name":96,"slug":97,"count":98,"latest_published_at":99},"Reviews","reviews",30,"2026-09-24T20:07:31.000Z",{"name":101,"slug":102,"count":103,"latest_published_at":104},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]