[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-new-benchmark-tests-llms-on-human-cognitive-skills-not-tasks":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},6933,"new-benchmark-tests-llms-on-human-cognitive-skills-not-tasks","New Benchmark Tests LLMs on Human Cognitive Skills, Not Tasks","A new benchmark built from neuropsychology tests shows LLMs and humans fail at different parts of the same tasks.","A new benchmark borrows tools from clinical psychology to find out what large language models actually can't do.\n\nResearchers built NeuroCognition from three adapted neuropsychological tests: Raven's Progressive Matrices for abstract relational reasoning, a spatial working memory task for goal-directed spatial updating, and the Wisconsin Card Sorting Test for cognitive flexibility. Running it across 156 models, they confirmed a general factor of capability that shows up consistently on standard leaderboards. But performance drops sharply once tasks involve images instead of text, and drops further as complexity increases. Compared against a human baseline, models and people don't fail in the same spots - they stumble on different parts of the same puzzles.\n\nThat mismatch is the real finding. Most LLM benchmarks measure whether a model finishes a task, not whether it reasons the way a person does, so a model can top an image-reasoning leaderboard while still lacking the basic spatial updating a person handles without thinking. The researchers also found that throwing more complex reasoning at these tasks doesn't reliably help - simple, human-like strategies sometimes work better, which cuts against the assumption that more chain-of-thought is always the fix.\n\nA model that aces trivia and still fumbles a test built for hospital patients is a good reminder that general capability and general intelligence are not the same claim.","[\"llm-evaluation\",\"cognitive-science\",\"benchmarks\",\"ai-research\"]","2026-09-18T04:00:00.000Z","2026-09-18T23:20:22.657Z","2026-09-18T23:20:34.591Z","published",null,[],"ai",[26,27,28,29],"llm-evaluation","cognitive-science","benchmarks","ai-research",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2603.02540",0,{"sections":36},[37,40,44,49,54,58,62,67,71,76,81,86,91,96],{"name":38,"slug":24,"count":39,"latest_published_at":18},"AI",4082,{"name":41,"slug":42,"count":43,"latest_published_at":18},"Security","security",661,{"name":45,"slug":46,"count":47,"latest_published_at":48},"Policy","policy",339,"2026-09-17T12:00:00.000Z",{"name":50,"slug":51,"count":52,"latest_published_at":53},"Deals","deals",179,"2026-06-29T20:02:07.000Z",{"name":55,"slug":56,"count":57,"latest_published_at":18},"Hardware","hardware",155,{"name":59,"slug":60,"count":61,"latest_published_at":18},"Science","science",125,{"name":63,"slug":64,"count":65,"latest_published_at":66},"Consumer Tech","consumer-tech",99,"2026-09-09T17:27:33.000Z",{"name":68,"slug":69,"count":70,"latest_published_at":18},"Dev Tools","dev-tools",78,{"name":72,"slug":73,"count":74,"latest_published_at":75},"Software","software",75,"2026-09-10T20:41:21.000Z",{"name":77,"slug":78,"count":79,"latest_published_at":80},"Startups","startups",55,"2026-09-09T23:14:29.000Z",{"name":82,"slug":83,"count":84,"latest_published_at":85},"Gaming","gaming",43,"2026-09-10T12:18:06.000Z",{"name":87,"slug":88,"count":89,"latest_published_at":90},"General","general",41,"2026-09-08T01:57:23.000Z",{"name":92,"slug":93,"count":94,"latest_published_at":95},"Reviews","reviews",20,"2026-06-24T12:00:01.000Z",{"name":97,"slug":98,"count":99,"latest_published_at":100},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]