[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-medical-ai-study-finds-accuracy-gains-hide-vision-regressions":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},8076,"medical-ai-study-finds-accuracy-gains-hide-vision-regressions","Medical AI Study Finds Accuracy Gains Hide Vision Regressions","A controlled audit of a medical vision-language model shows fine-tuning nudged answer accuracy up while the model actually leaned on images less, not more.","A new audit of medical AI training finds that accuracy scores and actual image use can move in opposite directions.\n\nResearchers ran a controlled study on Qwen2.5-VL-3B, a vision-language model, using the PMC-VQA medical visual-question-answering benchmark. They compared several post-training methods: supervised fine-tuning with LoRA restricted to the language model, broader multimodal adaptation, standard GRPO (a reinforcement-learning method that rewards correct answers), and a custom objective that specifically rewards using visual evidence. On 2,000 test questions, language-model-only LoRA fine-tuning nudged correct-image accuracy up by about 1.1 percentage points, but cases where the correct answer actually depended on seeing the image, so-called visual-benefit events, fell by 2.4 points, and image sensitivity dropped 5.6 points. Paired records showed the model gained real visual reasoning on 155 questions but lost it on 203 others, a net decline the aggregate accuracy number never revealed. Broader multimodal adaptation scored worse on correct-image accuracy, and standard GRPO produced inconsistent results tied to how mixed-reward groups get scored.\n\nAccuracy is the number hospitals, vendors, and regulators lean on to judge whether a medical model is improving, but this study shows accuracy can climb while the model quietly relies less on the actual image and more on guessable text patterns. That gap matters more in medicine than in a general chatbot: a model that answers convincingly without truly reading the scan is a liability, not a convenience.\n\nThe paper reads less like a broken model and more like a broken instrument. If a single accuracy score can hide whether a medical AI is seeing or guessing, every leaderboard built on that score needs a second column.","[\"medical ai\",\"vision-language models\",\"ai research\",\"ai benchmarks\"]","2026-09-28T04:00:00.000Z","2026-09-28T08:27:26.034Z","2026-09-28T08:27:32.511Z","published",null,[],"ai",[26,27,28,29],"medical ai","vision-language models","ai research","ai benchmarks",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2609.31450",0,{"sections":36},[37,40,44,49,54,59,63,68,73,78,83,88,92,97],{"name":38,"slug":24,"count":39,"latest_published_at":18},"AI",4792,{"name":41,"slug":42,"count":43,"latest_published_at":18},"Security","security",762,{"name":45,"slug":46,"count":47,"latest_published_at":48},"Policy","policy",399,"2026-09-27T18:39:02.000Z",{"name":50,"slug":51,"count":52,"latest_published_at":53},"Deals","deals",261,"2026-09-27T15:30:35.000Z",{"name":55,"slug":56,"count":57,"latest_published_at":58},"Hardware","hardware",188,"2026-09-27T20:46:36.000Z",{"name":60,"slug":61,"count":62,"latest_published_at":18},"Science","science",151,{"name":64,"slug":65,"count":66,"latest_published_at":67},"Consumer Tech","consumer-tech",135,"2026-09-26T14:30:00.000Z",{"name":69,"slug":70,"count":71,"latest_published_at":72},"Software","software",91,"2026-09-25T20:55:00.000Z",{"name":74,"slug":75,"count":76,"latest_published_at":77},"Dev Tools","dev-tools",84,"2026-09-26T04:20:58.000Z",{"name":79,"slug":80,"count":81,"latest_published_at":82},"Startups","startups",76,"2026-09-25T18:33:59.000Z",{"name":84,"slug":85,"count":86,"latest_published_at":87},"Gaming","gaming",48,"2026-09-25T18:35:21.000Z",{"name":89,"slug":90,"count":86,"latest_published_at":91},"General","general","2026-09-26T17:02:42.000Z",{"name":93,"slug":94,"count":95,"latest_published_at":96},"Reviews","reviews",30,"2026-09-24T20:07:31.000Z",{"name":98,"slug":99,"count":100,"latest_published_at":101},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]