[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-gradient-free-method-aims-to-fix-a-known-llm-alignment-bug":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},6662,"gradient-free-method-aims-to-fix-a-known-llm-alignment-bug","Gradient-Free Method Aims to Fix a Known LLM Alignment Bug","New research proposes a gradient-free method to fix likelihood displacement, a bug undermining today's dominant LLM preference-tuning technique.","A new alignment method skips the gradient math that trips up today's most popular way of tuning chatbots on human preferences.\n\nResearchers propose ComPO, short for Comparison-based Preference Optimization, a \"zeroth-order\" method - meaning it never computes a differentiable loss from a preference pair, and instead uses comparison oracles to figure out which direction to nudge the model. It targets a known flaw in Direct Preference Optimization (DPO): when two responses in a training pair are nearly equally likely, DPO can accidentally push down the probability of the preferred response too, a failure mode called likelihood displacement. The team built an offline version with a mathematical convergence proof and an online version that adds KL-based guardrails so the model does not drift too far from its starting point. They tested both on five open-source model families - Mistral, Llama, Gemma-2, Qwen3, and Gemma-3 - and reported better length-controlled win rates than existing direct alignment methods, backed by pair-level diagnostics.\n\nDPO and its variants are now the default way most labs fine-tune models on preference data, prized because they skip the separate reward model that older RLHF pipelines required. Likelihood displacement is a documented crack in that shortcut, so a gradient-free fix that avoids reintroducing a reward model would patch plumbing nearly every instruction-tuned model depends on.\n\nThe paper leans on convergence guarantees and diagnostics rather than leaderboard bragging, which is the right instinct this early - the real test is whether ComPO holds up outside five research checkpoints and inside an actual production fine-tuning run.","[\"llm-alignment\",\"preference-tuning\",\"ai-research\",\"arxiv\"]","2026-09-17T04:00:00.000Z","2026-09-18T04:56:29.505Z","2026-09-18T04:56:41.446Z","published",null,[],"ai",[26,27,28,29],"llm-alignment","preference-tuning","ai-research","arxiv",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2609.19144",0,{"sections":36},[37,41,45,50,55,59,63,68,73,77,82,87,92,97],{"name":38,"slug":24,"count":39,"latest_published_at":40},"AI",3852,"2026-09-17T08:27:09.000Z",{"name":42,"slug":43,"count":44,"latest_published_at":18},"Security","security",648,{"name":46,"slug":47,"count":48,"latest_published_at":49},"Policy","policy",338,"2026-09-11T04:00:00.000Z",{"name":51,"slug":52,"count":53,"latest_published_at":54},"Deals","deals",179,"2026-06-29T20:02:07.000Z",{"name":56,"slug":57,"count":58,"latest_published_at":18},"Hardware","hardware",154,{"name":60,"slug":61,"count":62,"latest_published_at":18},"Science","science",114,{"name":64,"slug":65,"count":66,"latest_published_at":67},"Consumer Tech","consumer-tech",99,"2026-09-09T17:27:33.000Z",{"name":69,"slug":70,"count":71,"latest_published_at":72},"Software","software",75,"2026-09-10T20:41:21.000Z",{"name":74,"slug":75,"count":76,"latest_published_at":18},"Dev Tools","dev-tools",73,{"name":78,"slug":79,"count":80,"latest_published_at":81},"Startups","startups",55,"2026-09-09T23:14:29.000Z",{"name":83,"slug":84,"count":85,"latest_published_at":86},"Gaming","gaming",43,"2026-09-10T12:18:06.000Z",{"name":88,"slug":89,"count":90,"latest_published_at":91},"General","general",41,"2026-09-08T01:57:23.000Z",{"name":93,"slug":94,"count":95,"latest_published_at":96},"Reviews","reviews",20,"2026-06-24T12:00:01.000Z",{"name":98,"slug":99,"count":100,"latest_published_at":101},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]