[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-new-dataset-exposes-how-ai-models-weigh-moral-tradeoffs":10,"sections":36},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":24,"persona_id":22,"persona_name":22,"section":25,"tags":26,"sources":31,"feedback":35,"feedback_at":22,"cost_usd":35,"total_tokens":35},7082,"new-dataset-exposes-how-ai-models-weigh-moral-tradeoffs","New Dataset Exposes How AI Models Weigh Moral Tradeoffs","A 12,000-question dataset shows language models have consistent, sometimes hidden leanings when honesty, justice, and autonomy collide.","Researchers built a 12,000-question test to catch language models quietly favoring one moral value over another.\n\nThe team created a dataset of two-option dilemmas covering three value conflicts: honesty versus justice, justice versus autonomy, and autonomy versus honesty. They translated the set into Hindi, Arabic, Spanish, and Chinese to see whether a model's answers hold up across languages. Testing GPT-5-mini with no explicit policy showed it consistently favored honesty over autonomy in all five languages. Meta's smaller Llama-3.2-1B and 3B models had a cruder problem: a strong bias toward whichever answer option came first, regardless of content. Both plain fine-tuning and Direct Preference Optimization eliminated that bias, pushing accuracy above 98 percent.\n\nThe more telling result came from a follow-up experiment with task vectors, the numerical traces of what fine-tuning changes inside a model. By isolating the direction tied to one specific value preference and separating it from the model's general instruction-following behavior, the researchers used task arithmetic to flip a model's stance to the opposite value entirely.\n\nA model whose ethics can be reversed with vector math is a useful research tool, but it is also a reminder that what looks like a considered moral position is often just a direction in weight space.","[\"ai ethics\",\"llm bias\",\"cross-lingual ai\",\"alignment research\"]","2026-09-21T04:00:00.000Z","2026-09-21T05:30:17.570Z","2026-09-21T05:30:29.126Z","published",null,[],"https:\u002F\u002Fcdn.xyz.onl\u002Farticle-images\u002Fnew-dataset-exposes-how-ai-models-weigh-moral-tradeoffs.webp","ai",[27,28,29,30],"ai ethics","llm bias","cross-lingual ai","alignment research",[32],{"name":33,"url":34},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2609.21094",0,{"sections":37},[38,41,45,50,55,60,65,70,75,80,85,90,95,100],{"name":39,"slug":25,"count":40,"latest_published_at":18},"AI",4158,{"name":42,"slug":43,"count":44,"latest_published_at":18},"Security","security",679,{"name":46,"slug":47,"count":48,"latest_published_at":49},"Policy","policy",350,"2026-09-20T20:32:43.000Z",{"name":51,"slug":52,"count":53,"latest_published_at":54},"Deals","deals",179,"2026-06-29T20:02:07.000Z",{"name":56,"slug":57,"count":58,"latest_published_at":59},"Hardware","hardware",156,"2026-09-19T11:00:00.000Z",{"name":61,"slug":62,"count":63,"latest_published_at":64},"Science","science",130,"2026-09-20T13:48:11.000Z",{"name":66,"slug":67,"count":68,"latest_published_at":69},"Consumer Tech","consumer-tech",99,"2026-09-09T17:27:33.000Z",{"name":71,"slug":72,"count":73,"latest_published_at":74},"Dev Tools","dev-tools",78,"2026-09-18T04:00:00.000Z",{"name":76,"slug":77,"count":78,"latest_published_at":79},"Software","software",75,"2026-09-10T20:41:21.000Z",{"name":81,"slug":82,"count":83,"latest_published_at":84},"Startups","startups",55,"2026-09-09T23:14:29.000Z",{"name":86,"slug":87,"count":88,"latest_published_at":89},"Gaming","gaming",43,"2026-09-10T12:18:06.000Z",{"name":91,"slug":92,"count":93,"latest_published_at":94},"General","general",42,"2026-09-18T22:35:10.000Z",{"name":96,"slug":97,"count":98,"latest_published_at":99},"Reviews","reviews",20,"2026-06-24T12:00:01.000Z",{"name":101,"slug":102,"count":103,"latest_published_at":104},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]