[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-study-finds-ai-reasoning-probes-may-grade-the-wrong-thing":10,"sections":40},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":30,"tags":31,"sources":35,"feedback":39,"feedback_at":22,"cost_usd":39,"total_tokens":39},8473,"study-finds-ai-reasoning-probes-may-grade-the-wrong-thing","Study Finds AI Reasoning Probes May Grade the Wrong Thing","A new arXiv paper says AI reasoning-quality probes often just measure question difficulty, and offers a same-question comparison method instead.","A new preprint suggests the tools researchers use to judge whether an AI's step-by-step reasoning is actually good have been measuring something else: how hard the question was, not how sound the reasoning is.\n\nThe paper, \"Rethinking Reasoning Paths as Phase-Structured Trajectories\" (arXiv:2609.36461), posted this week, looks at how researchers evaluate the hidden reasoning chains large language models generate before answering a question. The standard method tags every step in a chain with the final answer's correctness and trains a probe on that label across a mixed batch of different questions. The authors argue this lets the probe cheat: it can learn to spot which question is being asked rather than judge the actual quality of the reasoning path. Their fix, called PAIR (Phase-Aligned Intra-question Reasoning), instead compares multiple attempts at the same question, lines up each attempt's steps by relative progress instead of raw step count, and scores quality only against other attempts at that identical question.\n\nThe difference matters. When the researchers tested standard probes within a single question instead of across a mixed batch, accuracy dropped sharply, evidence those probes had been leaning on question-level shortcuts all along. PAIR's phase-specific signal held up better, and using it to rank or steer reasoning attempts changed which answers a model actually produced, not just how the attempts were scored.\n\nIt's one preprint, not yet peer reviewed, and the gains are shown on benchmark tasks rather than shipped products, but it's a useful check on how many \"the AI is reasoning well\" claims rest on evaluation methods nobody has stress-tested.","[\"ai\",\"llm-research\",\"interpretability\",\"arxiv\"]","2026-09-30T04:00:00.000Z","2026-09-30T05:53:58.016Z","2026-09-30T05:54:03.939Z","published",null,[24],{"id":25,"reviewer":26,"round":27,"reason":28,"status":29},"editor-r1","editor",1,"Add explicit attribution—the paper's title, arXiv ID, and posting date (arXiv:2609.36461, posted this week)—since the draft describes 'researchers' and their findings without ever naming the source publication or date needed to verify the story.","resolved","ai",[30,32,33,34],"llm-research","interpretability","arxiv",[36],{"name":37,"url":38},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2609.36461",0,{"sections":41},[42,45,49,53,58,63,68,73,78,83,88,93,98,103],{"name":43,"slug":30,"count":44,"latest_published_at":18},"AI",5028,{"name":46,"slug":47,"count":48,"latest_published_at":18},"Security","security",780,{"name":50,"slug":51,"count":52,"latest_published_at":18},"Policy","policy",417,{"name":54,"slug":55,"count":56,"latest_published_at":57},"Deals","deals",284,"2026-09-29T21:00:00.000Z",{"name":59,"slug":60,"count":61,"latest_published_at":62},"Hardware","hardware",194,"2026-09-29T13:16:04.000Z",{"name":64,"slug":65,"count":66,"latest_published_at":67},"Science","science",154,"2026-09-28T13:19:18.000Z",{"name":69,"slug":70,"count":71,"latest_published_at":72},"Consumer Tech","consumer-tech",142,"2026-09-29T18:38:03.000Z",{"name":74,"slug":75,"count":76,"latest_published_at":77},"Software","software",91,"2026-09-25T20:55:00.000Z",{"name":79,"slug":80,"count":81,"latest_published_at":82},"Dev Tools","dev-tools",89,"2026-09-29T17:15:00.000Z",{"name":84,"slug":85,"count":86,"latest_published_at":87},"Startups","startups",83,"2026-09-29T21:51:36.000Z",{"name":89,"slug":90,"count":91,"latest_published_at":92},"General","general",49,"2026-09-28T16:44:57.000Z",{"name":94,"slug":95,"count":96,"latest_published_at":97},"Gaming","gaming",48,"2026-09-25T18:35:21.000Z",{"name":99,"slug":100,"count":101,"latest_published_at":102},"Reviews","reviews",31,"2026-09-28T14:31:34.000Z",{"name":104,"slug":105,"count":106,"latest_published_at":107},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]