[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-new-test-shows-ai-math-reasoning-falters-across-languages":10,"sections":48},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":38,"tags":39,"sources":43,"feedback":47,"feedback_at":22,"cost_usd":47,"total_tokens":47},8701,"new-test-shows-ai-math-reasoning-falters-across-languages","New Test Shows AI Math Reasoning Falters Across Languages","A new benchmark reveals many AI models solve the same math problem inconsistently once names, digits, and language change.","AI models that ace grade-school math often stumble when the same problem shows up in a different language or with swapped numbers.\n\nResearchers behind the paper \"MGSM-Pro: A Simple Strategy for Robust Multilingual Mathematical Reasoning Evaluation\" (arXiv:2601.21225) built a tougher version of MGSM, an existing multilingual math benchmark. They generated five variants of each question, swapping names, digits, and irrelevant context, then tested models across nine languages. Low-resource languages saw sharp accuracy drops on digit swaps, even when the same models stayed steady in high-resource languages. The pattern held for proprietary systems too: Gemini 2.5 Flash and GPT-4.1 both lost ground on digit changes, while Gemini 3.0 Pro and open models GPT-OSS 120B and DeepSeek v3 proved more consistent.\n\nThat distinction matters because most benchmark leaderboards report a single score per language, not five. If changing a number from 12 to 47 flips a model's answer, it wasn't reasoning about the problem in the first place; it was pattern-matching to digits it had memorized. The researchers' fix is blunt but sensible: test each question with at least five digit variations before trusting the score.\n\nIt's a multilingual echo of an English-only finding from GSM-Symbolic, which found similar instability when questions were reworded. Robustness, it turns out, doesn't travel well between languages any more than it travels between phrasings.","[\"ai\",\"llm-benchmarks\",\"multilingual-nlp\",\"math-reasoning\"]","2026-09-30T04:00:00.000Z","2026-09-30T20:17:04.421Z","2026-09-30T20:17:10.636Z","published",null,[24,30,34],{"id":25,"reviewer":26,"round":27,"reason":28,"status":29},"editor-r1","editor",1,"Add explicit attribution (arXiv paper title\u002FID and posting date) since the draft never names its source, and replace vague performance descriptors like 'large performance drops' and 'held up better' with the actual accuracy\u002Fpercentage figures the comparison depends on.","resolved",{"id":31,"reviewer":26,"round":32,"reason":33,"status":29},"editor-r2",2,"The source abstract contains no actual accuracy\u002Fpercentage figures at all, so replace 'accuracy fall,' 'lost more ground,' and 'matched or beat' with real numbers pulled from the paper's results tables\u002Fappendix (not just the abstract) or explicitly flag that the abstract itself withholds numeric figures, and drop or verify the unsourced 'posted September 30, 2026' date, which doesn't appear anywhere in the provided source material and suspiciously matches today's date.",{"id":35,"reviewer":26,"round":36,"reason":37,"status":29},"editor-r3",3,"Add explicit attribution in the body naming the paper title and arXiv ID (arXiv:2601.21225, 'MGSM-Pro: A Simple Strategy for Robust Multilingual Mathematical Reasoning Evaluation') since the draft still never cites its source, and do not invent a posting date since none appears in the source material.","ai",[38,40,41,42],"llm-benchmarks","multilingual-nlp","math-reasoning",[44],{"name":45,"url":46},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2601.21225",0,{"sections":49},[50,53,57,61,66,71,75,80,85,89,94,99,104,109],{"name":51,"slug":38,"count":52,"latest_published_at":18},"AI",5184,{"name":54,"slug":55,"count":56,"latest_published_at":18},"Security","security",791,{"name":58,"slug":59,"count":60,"latest_published_at":18},"Policy","policy",417,{"name":62,"slug":63,"count":64,"latest_published_at":65},"Deals","deals",284,"2026-09-29T21:00:00.000Z",{"name":67,"slug":68,"count":69,"latest_published_at":70},"Hardware","hardware",194,"2026-09-29T13:16:04.000Z",{"name":72,"slug":73,"count":74,"latest_published_at":18},"Science","science",155,{"name":76,"slug":77,"count":78,"latest_published_at":79},"Consumer Tech","consumer-tech",142,"2026-09-29T18:38:03.000Z",{"name":81,"slug":82,"count":83,"latest_published_at":84},"Software","software",91,"2026-09-25T20:55:00.000Z",{"name":86,"slug":87,"count":88,"latest_published_at":18},"Dev Tools","dev-tools",90,{"name":90,"slug":91,"count":92,"latest_published_at":93},"Startups","startups",83,"2026-09-29T21:51:36.000Z",{"name":95,"slug":96,"count":97,"latest_published_at":98},"General","general",49,"2026-09-28T16:44:57.000Z",{"name":100,"slug":101,"count":102,"latest_published_at":103},"Gaming","gaming",48,"2026-09-25T18:35:21.000Z",{"name":105,"slug":106,"count":107,"latest_published_at":108},"Reviews","reviews",31,"2026-09-28T14:31:34.000Z",{"name":110,"slug":111,"count":112,"latest_published_at":113},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]