[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-new-method-lets-ai-models-shift-values-without-breaking-answers":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},8953,"new-method-lets-ai-models-shift-values-without-breaking-answers","New Method Lets AI Models Shift Values Without Breaking Answers","A new technique edits an AI model's values while preserving its facts and reasoning, beating a prompting baseline in early tests on LLaMA-3.1-8B.","Researchers have found a way to change what an AI model values without scrambling the facts it's working with.\n\nThe team built what they call an editable semantic-value interface: a layer added onto a frozen model's existing internal representations that separates \"what the facts are\" from \"how the model judges them.\" A one-way connection feeds semantic information into the value code, but a stop-gradient blocks that channel from feeding back and corrupting the facts when values get edited. Tested on two instruction-tuned backbones, the approach held onto more of the original meaning (BERTScore of 0.938 versus 0.923 for a prompting baseline) and produced fewer contradictions (5.1% versus 7.6%), while matching that baseline's alignment score (0.750 versus 0.748) on LLaMA-3.1-8B. A separate ablation isolated the representation learning from the edit mechanism itself, to confirm the gains come from the architecture and not just better prompting.\n\nThis matters because most value-steering techniques today are blunt instruments. Push a model toward \"more cautious\" and it starts refusing harmless requests or hedging on settled facts. Push it toward \"more helpful\" and it can bend the truth to please, which is a real problem for anyone deploying chat models at scale where safety tuning and usefulness are usually traded off against each other.\n\nTreat this as early lab results, not a shipped fix. It's a single arXiv preprint, tested on two backbones, with human ratings used only as supporting evidence rather than the primary measure. The gap over the prompting baseline is real but modest: a promising direction, not a solved problem.","[\"ai alignment\",\"llm steering\",\"interpretability\",\"arxiv\"]","2026-10-01T04:00:00.000Z","2026-10-01T12:34:34.348Z","2026-10-01T12:34:36.698Z","published",null,[],"ai",[26,27,28,29],"ai alignment","llm steering","interpretability","arxiv",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2609.39701",0,{"sections":36},[37,40,44,49,54,59,63,68,73,77,82,87,92,97],{"name":38,"slug":24,"count":39,"latest_published_at":18},"AI",5453,{"name":41,"slug":42,"count":43,"latest_published_at":18},"Security","security",805,{"name":45,"slug":46,"count":47,"latest_published_at":48},"Policy","policy",429,"2026-10-01T02:26:17.000Z",{"name":50,"slug":51,"count":52,"latest_published_at":53},"Deals","deals",298,"2026-09-30T21:00:26.000Z",{"name":55,"slug":56,"count":57,"latest_published_at":58},"Hardware","hardware",196,"2026-09-30T13:00:00.000Z",{"name":60,"slug":61,"count":62,"latest_published_at":18},"Science","science",159,{"name":64,"slug":65,"count":66,"latest_published_at":67},"Consumer Tech","consumer-tech",149,"2026-09-30T22:57:11.000Z",{"name":69,"slug":70,"count":71,"latest_published_at":72},"Dev Tools","dev-tools",93,"2026-10-01T02:30:48.000Z",{"name":74,"slug":75,"count":71,"latest_published_at":76},"Software","software","2026-09-30T21:41:11.000Z",{"name":78,"slug":79,"count":80,"latest_published_at":81},"Startups","startups",84,"2026-09-30T20:39:09.000Z",{"name":83,"slug":84,"count":85,"latest_published_at":86},"Gaming","gaming",51,"2026-09-30T16:24:30.000Z",{"name":88,"slug":89,"count":90,"latest_published_at":91},"General","general",50,"2026-09-30T21:37:54.000Z",{"name":93,"slug":94,"count":95,"latest_published_at":96},"Reviews","reviews",31,"2026-09-28T14:31:34.000Z",{"name":98,"slug":99,"count":100,"latest_published_at":101},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]