[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-new-framework-grades-context-in-health-chat-privacy-risks":10,"sections":40},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":30,"tags":31,"sources":35,"feedback":39,"feedback_at":22,"cost_usd":39,"total_tokens":39},9273,"new-framework-grades-context-in-health-chat-privacy-risks","New Framework Grades Context in Health Chat Privacy Risks","A new evaluation framework tests whether AI models can tell confirmed, suspected, or negated health conditions apart in online medical chats.","A new framework grades how risky a piece of health information really is based on context, not just the words used.\n\nResearchers built a framework for evaluating sensitivity in online medical conversations, like forum posts or chat-based consultations, that goes beyond simple entity labeling. It tracks four context signals drawn from clinical documentation standards: whether a condition is confirmed, suspected, negated, or hypothetical; who the symptom belongs to; whether a related test result is pending or returned; and how specific the disclosure is. The team built contrastive test cases - near-identical conversation snippets that alter just one of those signals at a time - and used them to compare large language models that see only a bare mention of a condition against models given the full surrounding context.\n\nPrivacy tools that flag sensitive health disclosures are only as good as their ability to separate a real diagnosis from a suspected, negated, or hypothetical one. Mislabeling a hypothetical or negated condition as confirmed could trigger unnecessary alerts or over-redaction, while missing a genuinely confirmed disclosure could let sensitive data slip through unflagged. The framework is built to measure both of those failure modes separately, rather than folding them into one accuracy score.\n\nIt is a reminder that most AI classification benchmarks test what was said, not what was meant - and in a medical chat log, that distinction is the whole point.","[\"ai\",\"privacy\",\"health-data\",\"llm-evaluation\"]","2026-10-01T04:00:00.000Z","2026-10-02T06:20:42.087Z","2026-10-02T06:20:45.892Z","published",null,[24],{"id":25,"reviewer":26,"round":27,"reason":28,"status":29},"editor-r1","editor",1,"The source abstract only describes the framework's design and goals ('aims to quantify... characterize errors') with no stated results, so the headline's claim that the study 'Finds LLMs Struggle' is an unsupported finding — rewrite the headline\u002Fdek to describe the framework and its purpose rather than asserting a result, and drop the invented 'confirmed cancer diagnosis' example since cancer is never mentioned in the source.","resolved","ai",[30,32,33,34],"privacy","health-data","llm-evaluation",[36],{"name":37,"url":38},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2601.09717",0,{"sections":41},[42,45,49,53,58,63,67,72,77,81,86,91,96,101],{"name":43,"slug":30,"count":44,"latest_published_at":18},"AI",5671,{"name":46,"slug":47,"count":48,"latest_published_at":18},"Security","security",820,{"name":50,"slug":51,"count":52,"latest_published_at":18},"Policy","policy",430,{"name":54,"slug":55,"count":56,"latest_published_at":57},"Deals","deals",298,"2026-09-30T21:00:26.000Z",{"name":59,"slug":60,"count":61,"latest_published_at":62},"Hardware","hardware",196,"2026-09-30T13:00:00.000Z",{"name":64,"slug":65,"count":66,"latest_published_at":18},"Science","science",165,{"name":68,"slug":69,"count":70,"latest_published_at":71},"Consumer Tech","consumer-tech",149,"2026-09-30T22:57:11.000Z",{"name":73,"slug":74,"count":75,"latest_published_at":76},"Dev Tools","dev-tools",93,"2026-10-01T02:30:48.000Z",{"name":78,"slug":79,"count":75,"latest_published_at":80},"Software","software","2026-09-30T21:41:11.000Z",{"name":82,"slug":83,"count":84,"latest_published_at":85},"Startups","startups",84,"2026-09-30T20:39:09.000Z",{"name":87,"slug":88,"count":89,"latest_published_at":90},"Gaming","gaming",51,"2026-09-30T16:24:30.000Z",{"name":92,"slug":93,"count":94,"latest_published_at":95},"General","general",50,"2026-09-30T21:37:54.000Z",{"name":97,"slug":98,"count":99,"latest_published_at":100},"Reviews","reviews",31,"2026-09-28T14:31:34.000Z",{"name":102,"slug":103,"count":104,"latest_published_at":105},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]