[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-llms-reason-better-about-math-mistakes-than-science-ones":10,"sections":34},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":29,"feedback":33,"feedback_at":22,"cost_usd":33,"total_tokens":33},6461,"llms-reason-better-about-math-mistakes-than-science-ones","LLMs Reason Better About Math Mistakes Than Science Ones","A new study finds AI models produce plausible wrong answers for math questions by simulating student errors, but rely on shakier tricks for science questions.","Researchers tested whether large language models can convincingly fake being wrong.\n\nThe task was distractor generation: writing the incorrect-but-plausible answer choices that fill out a multiple-choice question. Researchers built a taxonomy of reasoning strategies for this task, drawn from learning-science research, then applied it to LLM-generated reasoning traces on math and science question sets. On math problems, models followed a coherent process: solve the problem correctly, articulate a specific student misconception, simulate the error that misconception would produce, then pick a plausible wrong answer from the result. On science problems, models leaned instead on surface-level semantic similarity to the right answer, a shakier substitute for actually modeling why a student would get it wrong. The most common failures were models failing to solve the problem correctly in the first place, or generating good distractor candidates and then discarding them during selection. Feeding the model the correct solution upfront improved alignment with human-authored distractors by 6.4%.\n\nThis matters because distractor quality is a real bottleneck in ed-tech, not an academic curiosity. Bad multiple-choice options either give the answer away or fail to catch anything useful about what a student actually misunderstands, and human item-writers are expensive and slow. A model that can simulate specific misconceptions rather than pattern-match on similarity is a meaningfully better tool for building diagnostic assessments, not just filler quizzes.\n\nThe gap between math and science performance is the real finding here: it suggests these models reason well when a domain has clean, checkable logical steps to break, and reason worse when the \"error\" is more about conceptual fuzziness than a traceable calculation gone wrong.","[\"ai\",\"education\",\"llm-research\",\"arxiv\"]","2026-09-16T04:00:00.000Z","2026-09-17T17:47:36.340Z","2026-09-17T17:47:48.339Z","published",null,[],"ai",[24,26,27,28],"education","llm-research","arxiv",[30],{"name":31,"url":32},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2603.15547",0,{"sections":35},[36,40,45,50,55,59,63,68,73,77,82,87,92,97],{"name":37,"slug":24,"count":38,"latest_published_at":39},"AI",3852,"2026-09-17T08:27:09.000Z",{"name":41,"slug":42,"count":43,"latest_published_at":44},"Security","security",648,"2026-09-17T04:00:00.000Z",{"name":46,"slug":47,"count":48,"latest_published_at":49},"Policy","policy",338,"2026-09-11T04:00:00.000Z",{"name":51,"slug":52,"count":53,"latest_published_at":54},"Deals","deals",179,"2026-06-29T20:02:07.000Z",{"name":56,"slug":57,"count":58,"latest_published_at":44},"Hardware","hardware",154,{"name":60,"slug":61,"count":62,"latest_published_at":44},"Science","science",114,{"name":64,"slug":65,"count":66,"latest_published_at":67},"Consumer Tech","consumer-tech",99,"2026-09-09T17:27:33.000Z",{"name":69,"slug":70,"count":71,"latest_published_at":72},"Software","software",75,"2026-09-10T20:41:21.000Z",{"name":74,"slug":75,"count":76,"latest_published_at":44},"Dev Tools","dev-tools",73,{"name":78,"slug":79,"count":80,"latest_published_at":81},"Startups","startups",55,"2026-09-09T23:14:29.000Z",{"name":83,"slug":84,"count":85,"latest_published_at":86},"Gaming","gaming",43,"2026-09-10T12:18:06.000Z",{"name":88,"slug":89,"count":90,"latest_published_at":91},"General","general",41,"2026-09-08T01:57:23.000Z",{"name":93,"slug":94,"count":95,"latest_published_at":96},"Reviews","reviews",20,"2026-06-24T12:00:01.000Z",{"name":98,"slug":99,"count":100,"latest_published_at":101},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]