[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-chatbots-trust-your-language-even-when-its-wikipedia-is-wrong":10,"sections":34},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":29,"feedback":33,"feedback_at":22,"cost_usd":33,"total_tokens":33},9651,"chatbots-trust-your-language-even-when-its-wikipedia-is-wrong","Chatbots Trust Your Language Even When Its Wikipedia Is Wrong","A new benchmark shows large language models give different answers to the same question depending on the query language, when sources conflict.","Ask an AI chatbot a factual question, and the language you ask in can quietly change the answer you get. The model won't flag the discrepancy.\n\nResearchers built a benchmark called Waldo, drawn from 12,000 question-answer pairs sourced from Wikipedia, to test how large language models handle cross-lingual knowledge gaps and conflicts. They evaluated eight models across five languages. When one language's Wikipedia edition simply lacked a fact, models generally pulled the answer from whichever language did have it, regardless of the question's language. But when two editions gave conflicting versions of the same fact, the models overwhelmingly sided with the source written in the query's own language.\n\nThat split behavior matters because it means two people asking an identical question in different languages can get contradictory facts, with no signal that a dispute exists. For topics like history, geography, or politics, where Wikipedia editions diverge most, that quietly locks users into a language-specific version of events.\n\nThe researchers tested two fixes, including fine-tuning with LoRA, which closed up to 61.5 percent of the gap, proof that the bias is addressable but not solved. So much for \"multilingual\" meaning neutral.","[\"ai\",\"multilingual-ai\",\"wikipedia\",\"llm-bias\"]","2026-10-02T04:00:00.000Z","2026-10-03T04:38:31.247Z","2026-10-03T04:38:37.450Z","published",null,[],"ai",[24,26,27,28],"multilingual-ai","wikipedia","llm-bias",[30],{"name":31,"url":32},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2610.00606",0,{"sections":35},[36,39,43,47,52,56,60,65,70,75,80,85,90,95],{"name":37,"slug":24,"count":38,"latest_published_at":18},"AI",5976,{"name":40,"slug":41,"count":42,"latest_published_at":18},"Security","security",842,{"name":44,"slug":45,"count":46,"latest_published_at":18},"Policy","policy",438,{"name":48,"slug":49,"count":50,"latest_published_at":51},"Deals","deals",317,"2026-10-01T22:00:00.000Z",{"name":53,"slug":54,"count":55,"latest_published_at":18},"Hardware","hardware",199,{"name":57,"slug":58,"count":59,"latest_published_at":18},"Science","science",173,{"name":61,"slug":62,"count":63,"latest_published_at":64},"Consumer Tech","consumer-tech",155,"2026-10-01T19:54:10.000Z",{"name":66,"slug":67,"count":68,"latest_published_at":69},"Dev Tools","dev-tools",96,"2026-10-01T16:57:03.000Z",{"name":71,"slug":72,"count":73,"latest_published_at":74},"Software","software",93,"2026-09-30T21:41:11.000Z",{"name":76,"slug":77,"count":78,"latest_published_at":79},"Startups","startups",90,"2026-10-01T21:55:22.000Z",{"name":81,"slug":82,"count":83,"latest_published_at":84},"Gaming","gaming",53,"2026-10-02T02:50:39.000Z",{"name":86,"slug":87,"count":88,"latest_published_at":89},"General","general",50,"2026-09-30T21:37:54.000Z",{"name":91,"slug":92,"count":93,"latest_published_at":94},"Reviews","reviews",31,"2026-09-28T14:31:34.000Z",{"name":96,"slug":97,"count":98,"latest_published_at":99},"How-To","how-to",7,"2026-10-01T09:00:00.000Z"]