[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-study-finds-llms-organize-math-reasoning-by-method-not-topic":10,"sections":34},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":29,"feedback":33,"feedback_at":22,"cost_usd":33,"total_tokens":33},7534,"study-finds-llms-organize-math-reasoning-by-method-not-topic","Study Finds LLMs Organize Math Reasoning by Method, Not Topic","New research shows math-capable LLMs cluster their internal reasoning by method, not topic, which could upend how benchmarks and training data are built.","When a large language model solves a math problem, it turns out the model is not thinking in terms of \"algebra\" or \"geometry.\" It is thinking in terms of technique.\n\nResearchers tested eight open math-capable models against five reasoning benchmarks using a generation-replay protocol: have a model solve a problem, replay its own prompt-plus-solution trajectory, then extract activation-importance signatures from the reasoning tokens. Clustering those signatures without supervision produced groups that beat random baselines in all 40 model-source combinations tested. Two independent frontier-LLM judges rated 77 to 82 percent of the resulting clusters as coherent by approach, versus just 6 to 11 percent for control clusters built from the same topic. When researchers explicitly asked models to use a different reasoning approach, cluster assignment shifted in seven of eight model conditions, while simply rephrasing a question left it unchanged.\n\nThis matters because nearly every math benchmark in wide use, from GSM8K to MATH, is organized by topic: algebra, geometry, word problems. If models are actually routing computation by method instead, a benchmark or training set that is carefully balanced across topics can still be badly skewed across reasoning approaches without anyone noticing. That is a blind spot baked into how the field measures and trains math ability.\n\nIt is a reminder that the categories humans find intuitive for organizing a subject are not necessarily the categories a model builds for itself.","[\"ai\",\"llms\",\"math-reasoning\",\"benchmarks\"]","2026-09-24T04:00:00.000Z","2026-09-24T04:58:41.563Z","2026-09-24T04:58:46.646Z","published",null,[],"ai",[24,26,27,28],"llms","math-reasoning","benchmarks",[30],{"name":31,"url":32},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2609.27041",0,{"sections":35},[36,39,43,48,53,58,63,68,73,78,83,88,93,98],{"name":37,"slug":24,"count":38,"latest_published_at":18},"AI",4387,{"name":40,"slug":41,"count":42,"latest_published_at":18},"Security","security",720,{"name":44,"slug":45,"count":46,"latest_published_at":47},"Policy","policy",380,"2026-09-23T22:53:43.000Z",{"name":49,"slug":50,"count":51,"latest_published_at":52},"Deals","deals",220,"2026-09-23T22:00:00.000Z",{"name":54,"slug":55,"count":56,"latest_published_at":57},"Hardware","hardware",173,"2026-09-23T23:42:10.000Z",{"name":59,"slug":60,"count":61,"latest_published_at":62},"Science","science",135,"2026-09-23T22:45:49.000Z",{"name":64,"slug":65,"count":66,"latest_published_at":67},"Consumer Tech","consumer-tech",116,"2026-09-24T00:51:49.000Z",{"name":69,"slug":70,"count":71,"latest_published_at":72},"Software","software",85,"2026-09-23T20:00:00.000Z",{"name":74,"slug":75,"count":76,"latest_published_at":77},"Dev Tools","dev-tools",79,"2026-09-22T22:21:13.000Z",{"name":79,"slug":80,"count":81,"latest_published_at":82},"Startups","startups",66,"2026-09-23T17:28:38.000Z",{"name":84,"slug":85,"count":86,"latest_published_at":87},"Gaming","gaming",45,"2026-09-22T15:35:06.000Z",{"name":89,"slug":90,"count":91,"latest_published_at":92},"General","general",43,"2026-09-21T23:48:56.000Z",{"name":94,"slug":95,"count":96,"latest_published_at":97},"Reviews","reviews",27,"2026-09-22T13:00:00.000Z",{"name":99,"slug":100,"count":101,"latest_published_at":102},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]