[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-openai-ships-a-benchmark-for-mental-health-ai-chats":10,"sections":40},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":30,"tags":31,"sources":35,"feedback":39,"feedback_at":22,"cost_usd":39,"total_tokens":39},7493,"openai-ships-a-benchmark-for-mental-health-ai-chats","OpenAI Ships a Benchmark for Mental-Health AI Chats","OpenAI's new benchmark grades chatbot responses to realistic mental-health conversations, checking for both helpfulness and safety.","OpenAI has released a benchmark for grading how AI chatbots handle mental-health conversations.\n\nThe new tool, called MentalHealthBench, was built with input from mental-health experts. It scores AI responses to realistic mental-health conversations on two axes: how helpful they are and how safe they are. OpenAI frames the benchmark as a way to measure whether a model's answers actually help someone in distress, not just whether they sound reasonable. The announcement does not detail how many conversations the benchmark covers or how the expert scoring was calibrated.\n\nThis matters because chatbots have quietly become a front line for people seeking emotional support, whether or not they were built for that job. A standardized, expert-informed yardstick gives AI labs, and outside researchers, something concrete to test against instead of relying on scattered anecdotes about chatbots handling a crisis well or badly.\n\nA benchmark is easy to publish and hard to verify; the real test is whether MentalHealthBench holds up once researchers outside OpenAI start probing it.","[\"ai\",\"mental-health\",\"benchmarks\",\"openai\"]","2026-09-23T10:00:00.000Z","2026-09-23T20:54:09.064Z","2026-09-23T20:54:13.278Z","published",null,[24],{"id":25,"reviewer":26,"round":27,"reason":28,"status":29},"editor-r1","editor",1,"Remove or verify the claim that the benchmark evaluates 'multi-turn' conversations versus 'isolated question-and-answer pairs' — the source only says 'realistic mental health conversations' and does not mention conversation structure, so this is an invented specific not supported by the source material.","resolved","ai",[30,32,33,34],"mental-health","benchmarks","openai",[36],{"name":37,"url":38},"OpenAI","https:\u002F\u002Fopenai.com\u002Findex\u002Fintroducing-mentalhealthbench",0,{"sections":41},[42,46,51,56,61,66,71,76,81,86,91,96,101,106],{"name":43,"slug":30,"count":44,"latest_published_at":45},"AI",4363,"2026-09-23T23:03:26.000Z",{"name":47,"slug":48,"count":49,"latest_published_at":50},"Security","security",718,"2026-09-23T20:24:22.000Z",{"name":52,"slug":53,"count":54,"latest_published_at":55},"Policy","policy",380,"2026-09-23T22:53:43.000Z",{"name":57,"slug":58,"count":59,"latest_published_at":60},"Deals","deals",220,"2026-09-23T22:00:00.000Z",{"name":62,"slug":63,"count":64,"latest_published_at":65},"Hardware","hardware",172,"2026-09-23T20:44:59.000Z",{"name":67,"slug":68,"count":69,"latest_published_at":70},"Science","science",134,"2026-09-23T09:00:00.000Z",{"name":72,"slug":73,"count":74,"latest_published_at":75},"Consumer Tech","consumer-tech",111,"2026-09-23T17:04:48.000Z",{"name":77,"slug":78,"count":79,"latest_published_at":80},"Software","software",85,"2026-09-23T20:00:00.000Z",{"name":82,"slug":83,"count":84,"latest_published_at":85},"Dev Tools","dev-tools",79,"2026-09-22T22:21:13.000Z",{"name":87,"slug":88,"count":89,"latest_published_at":90},"Startups","startups",66,"2026-09-23T17:28:38.000Z",{"name":92,"slug":93,"count":94,"latest_published_at":95},"Gaming","gaming",45,"2026-09-22T15:35:06.000Z",{"name":97,"slug":98,"count":99,"latest_published_at":100},"General","general",43,"2026-09-21T23:48:56.000Z",{"name":102,"slug":103,"count":104,"latest_published_at":105},"Reviews","reviews",27,"2026-09-22T13:00:00.000Z",{"name":107,"slug":108,"count":109,"latest_published_at":110},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]