[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-new-benchmark-tests-whether-ai-correctly-judges-medical-urgency":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},8103,"new-benchmark-tests-whether-ai-correctly-judges-medical-urgency","New Benchmark Tests Whether AI Correctly Judges Medical Urgency","AcuityBench, a new benchmark, finds top AI models still misjudge care urgency, especially in ambiguous cases.","A new benchmark shows leading AI models routinely misjudge how urgently a health complaint needs care, and the phrasing of the question changes which way they get it wrong.\n\nResearchers built AcuityBench by combining five public datasets, including user conversations, forum posts, clinical vignettes, and patient portal messages, into a single four-level urgency scale running from home monitoring to immediate emergency care. The benchmark includes 914 cases: 697 where experts agree on the right acuity level, and 217 cases physician reviewers flagged as genuinely ambiguous. Models were tested two ways: answering a direct multiple-choice question about urgency, and responding conversationally, with a rubric-based judge scoring the reply against the same four-level scale. Across 12 frontier proprietary and open-weight models, accuracy on the clear-cut cases varied widely, and models made different kinds of mistakes depending on which format they were tested in.\n\nConversational answers cut down on over-triage, where a model tells someone with a minor issue to rush to the ER, but they increased under-triage, where a serious symptom gets waved off, particularly in the highest-acuity cases. That's the more dangerous failure mode: a chatbot that talks a user out of urgent care is the one that could get someone hurt. In the ambiguous cases, no model's judgments matched the spread of physician opinion, and models sounded more confident than the disagreement among human experts actually warranted.\n\nHealth AI benchmarks have mostly measured whether a model can pass a medical licensing exam; this one measures something closer to bedside judgment, and the results suggest that gap hasn't closed.","[\"ai safety\",\"healthcare ai\",\"benchmarks\",\"llm evaluation\"]","2026-09-28T04:00:00.000Z","2026-09-28T09:36:35.789Z","2026-09-28T09:36:42.060Z","published",null,[],"ai",[26,27,28,29],"ai safety","healthcare ai","benchmarks","llm evaluation",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2605.11398",0,{"sections":36},[37,40,44,49,54,59,63,68,73,78,83,88,92,97],{"name":38,"slug":24,"count":39,"latest_published_at":18},"AI",4791,{"name":41,"slug":42,"count":43,"latest_published_at":18},"Security","security",762,{"name":45,"slug":46,"count":47,"latest_published_at":48},"Policy","policy",399,"2026-09-27T18:39:02.000Z",{"name":50,"slug":51,"count":52,"latest_published_at":53},"Deals","deals",261,"2026-09-27T15:30:35.000Z",{"name":55,"slug":56,"count":57,"latest_published_at":58},"Hardware","hardware",188,"2026-09-27T20:46:36.000Z",{"name":60,"slug":61,"count":62,"latest_published_at":18},"Science","science",151,{"name":64,"slug":65,"count":66,"latest_published_at":67},"Consumer Tech","consumer-tech",135,"2026-09-26T14:30:00.000Z",{"name":69,"slug":70,"count":71,"latest_published_at":72},"Software","software",91,"2026-09-25T20:55:00.000Z",{"name":74,"slug":75,"count":76,"latest_published_at":77},"Dev Tools","dev-tools",84,"2026-09-26T04:20:58.000Z",{"name":79,"slug":80,"count":81,"latest_published_at":82},"Startups","startups",76,"2026-09-25T18:33:59.000Z",{"name":84,"slug":85,"count":86,"latest_published_at":87},"Gaming","gaming",48,"2026-09-25T18:35:21.000Z",{"name":89,"slug":90,"count":86,"latest_published_at":91},"General","general","2026-09-26T17:02:42.000Z",{"name":93,"slug":94,"count":95,"latest_published_at":96},"Reviews","reviews",30,"2026-09-24T20:07:31.000Z",{"name":98,"slug":99,"count":100,"latest_published_at":101},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]