[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-bilingual-benchmark-tests-how-well-ai-catches-medical-errors":10,"sections":34},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":29,"feedback":33,"feedback_at":22,"cost_usd":33,"total_tokens":33},9263,"bilingual-benchmark-tests-how-well-ai-catches-medical-errors","Bilingual Benchmark Tests How Well AI Catches Medical Errors","A new bilingual benchmark finds general reasoning models beat medical-specialized ones at catching clinical errors, with scores varying by language.","Researchers just built a test to see whether AI can catch mistakes in doctors' notes - in two languages, not just English.\n\nThe benchmark, called MedRECT, pulls 663 error-laced passages from Japan's medical licensing exams and 458 more from the existing MEDEC dataset, then asks models to do three things: spot an error, identify which sentence has it, and fix it. The team ran 11 models, mixing proprietary and open-weight systems, through 17 different configurations, some with medical fine-tuning and some with extended reasoning turned on. Qwen3-32B did sharply better when reasoning was switched on, improving sentence-extraction accuracy by 24.5 percentage points on the Japanese set and 10.3 points on the English one. Several general-purpose reasoning models also beat all three medical-specialized models on detection and extraction.\n\nThat gap matters because hospitals shopping for AI scribes or chart-checking tools often assume a medical-tuned model will outperform a general one by default. This benchmark says that's not a safe bet, and that turning on a model's reasoning mode may do more than bolting on medical training data. It's also one of the few clinical-error datasets built outside English, which matters given how much deployed healthcare AI gets trained and graded on English text alone.\n\nLoRA fine-tuning did close some of the gap on correction accuracy, so specialization isn't worthless - it's just not the whole story the medical-AI pitch decks tell.","[\"ai\",\"llm benchmarks\",\"healthcare ai\",\"medical errors\"]","2026-10-01T04:00:00.000Z","2026-10-02T05:29:08.497Z","2026-10-02T05:29:12.152Z","published",null,[],"ai",[24,26,27,28],"llm benchmarks","healthcare ai","medical errors",[30],{"name":31,"url":32},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2511.00421",0,{"sections":35},[36,39,43,47,52,57,61,66,71,75,80,85,90,95],{"name":37,"slug":24,"count":38,"latest_published_at":18},"AI",5671,{"name":40,"slug":41,"count":42,"latest_published_at":18},"Security","security",820,{"name":44,"slug":45,"count":46,"latest_published_at":18},"Policy","policy",430,{"name":48,"slug":49,"count":50,"latest_published_at":51},"Deals","deals",298,"2026-09-30T21:00:26.000Z",{"name":53,"slug":54,"count":55,"latest_published_at":56},"Hardware","hardware",196,"2026-09-30T13:00:00.000Z",{"name":58,"slug":59,"count":60,"latest_published_at":18},"Science","science",165,{"name":62,"slug":63,"count":64,"latest_published_at":65},"Consumer Tech","consumer-tech",149,"2026-09-30T22:57:11.000Z",{"name":67,"slug":68,"count":69,"latest_published_at":70},"Dev Tools","dev-tools",93,"2026-10-01T02:30:48.000Z",{"name":72,"slug":73,"count":69,"latest_published_at":74},"Software","software","2026-09-30T21:41:11.000Z",{"name":76,"slug":77,"count":78,"latest_published_at":79},"Startups","startups",84,"2026-09-30T20:39:09.000Z",{"name":81,"slug":82,"count":83,"latest_published_at":84},"Gaming","gaming",51,"2026-09-30T16:24:30.000Z",{"name":86,"slug":87,"count":88,"latest_published_at":89},"General","general",50,"2026-09-30T21:37:54.000Z",{"name":91,"slug":92,"count":93,"latest_published_at":94},"Reviews","reviews",31,"2026-09-28T14:31:34.000Z",{"name":96,"slug":97,"count":98,"latest_published_at":99},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]