[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-japanese-riddles-expose-a-blind-spot-in-ai-reasoning":10,"sections":34},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":29,"feedback":33,"feedback_at":22,"cost_usd":33,"total_tokens":33},5071,"japanese-riddles-expose-a-blind-spot-in-ai-reasoning","Japanese riddles expose a blind spot in AI reasoning","A new benchmark built from kids' riddles shows top language models know the right answer more often than they're willing to say it.","A benchmark built from Japanese children's riddles just caught frontier AI models doing something odd: getting the right answer, then rejecting it.\n\nResearchers assembled 201 nazonazo, a genre of Japanese wordplay riddles that require reinterpreting a question rather than recalling a fact, then tested 38 frontier language models released between 2023 and 2025 under retrieval-free, zero-shot conditions, meaning no web lookups and no practice on the exact questions. They compared the models against 126 human solvers who averaged 52.9% accuracy on a 120-item subset. Non-reasoning models scored 7.6%. Models built for step-by-step reasoning did better, but still only reached 17.6%. The more telling finding came from reading the models' own reasoning transcripts: models frequently produced the correct answer as a candidate partway through their reasoning, then abandoned it for a wrong final answer.\n\nThat gap between generating a good idea and trusting it is the real story here. The researchers call it verification failure, and say it accounts for between 5% and 39% of a given model's wrong answers, depending on which model you look at. It suggests the constraint isn't creativity, it's judgment: these systems can stumble onto insight but lack a reliable way to recognize it as insight.\n\nThat's a more useful diagnosis than another leaderboard, especially given how many reasoning benchmarks these models ace partly because the answers leaked into training data somewhere along the way. A puzzle set that humans still only solve about half the time, and that can be refreshed indefinitely, is a harder thing to game - which may be exactly why the scores here look so much worse.","[\"ai\",\"ai benchmarks\",\"reasoning\",\"metacognition\"]","2026-08-17T04:00:00.000Z","2026-08-17T08:53:24.776Z","2026-08-17T08:53:36.589Z","published",null,[],"ai",[24,26,27,28],"ai benchmarks","reasoning","metacognition",[30],{"name":31,"url":32},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2509.14704",0,{"sections":35},[36,40,44,49,54,59,64,69,74,79,84,89,94,99],{"name":37,"slug":24,"count":38,"latest_published_at":39},"AI",3293,"2026-08-20T04:00:00.000Z",{"name":41,"slug":42,"count":43,"latest_published_at":39},"Security","security",435,{"name":45,"slug":46,"count":47,"latest_published_at":48},"Policy","policy",210,"2026-08-19T09:32:27.000Z",{"name":50,"slug":51,"count":52,"latest_published_at":53},"Deals","deals",179,"2026-06-29T20:02:07.000Z",{"name":55,"slug":56,"count":57,"latest_published_at":58},"Hardware","hardware",140,"2026-08-19T18:25:42.000Z",{"name":60,"slug":61,"count":62,"latest_published_at":63},"Consumer Tech","consumer-tech",95,"2026-08-18T16:05:00.000Z",{"name":65,"slug":66,"count":67,"latest_published_at":68},"Science","science",90,"2026-08-19T18:41:02.000Z",{"name":70,"slug":71,"count":72,"latest_published_at":73},"Software","software",73,"2026-08-18T07:51:50.000Z",{"name":75,"slug":76,"count":77,"latest_published_at":78},"Dev Tools","dev-tools",69,"2026-08-18T04:00:00.000Z",{"name":80,"slug":81,"count":82,"latest_published_at":83},"Startups","startups",47,"2026-08-19T19:13:46.000Z",{"name":85,"slug":86,"count":87,"latest_published_at":88},"Gaming","gaming",41,"2026-07-09T04:00:00.000Z",{"name":90,"slug":91,"count":92,"latest_published_at":93},"General","general",33,"2026-08-18T22:18:13.000Z",{"name":95,"slug":96,"count":97,"latest_published_at":98},"Reviews","reviews",20,"2026-06-24T12:00:01.000Z",{"name":100,"slug":101,"count":102,"latest_published_at":103},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]