[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-new-method-skips-useless-tokens-when-distilling-ai-models":10,"sections":34},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":29,"feedback":33,"feedback_at":22,"cost_usd":33,"total_tokens":33},7855,"new-method-skips-useless-tokens-when-distilling-ai-models","New Method Skips Useless Tokens When Distilling AI Models","A new technique masks low-value training signals so smaller AI models teach larger ones only where it actually matters, boosting accuracy on math benchmarks.","AI labs have started using a trick called Direct On-Policy Distillation to cheaply pass reinforcement-learning gains from a small, RL-trained model to a bigger student model, using the token-by-token probability shift between the small model's pre- and post-RL versions as a training signal. A new arXiv paper argues that trick is measuring the wrong thing: that shift can stay constant even when the actual behavioral difference between the two checkpoints has collapsed to near zero, meaning the student ends up training on tokens where the teacher barely changed at all. The researchers built a method, Selective Supervision for Direct-OPD (S2D-OPD), that ranks each token by how much the teacher's behavior actually diverged and keeps only the top 10 percent of that signal, discarding the rest. Tested on four student models between 1.7 billion and 8 billion parameters across two teacher pairs, the selective version beat plain Direct-OPD on the AIME and HMMT math benchmarks in seven of eight configurations, and tied it in the last one, without running any extra passes through the model.\n\nThat's a real efficiency win, not just a cleanup: distillation pipelines already burn compute on every token pushed through the student, so trimming supervision to the 10 percent that actually carries a teacher's behavior change means the same training run does more with the same budget. It also lands a small blow against the assumption, common across distillation work, that denser supervision is automatically better supervision.\n\nThe paper is a preprint with anonymized code, so treat the seven-of-eight scorecard as promising rather than settled until it survives peer review and gets tried outside math-competition benchmarks.","[\"ai\",\"distillation\",\"reinforcement-learning\",\"llm-research\"]","2026-09-25T04:00:00.000Z","2026-09-26T03:05:16.394Z","2026-09-26T03:05:20.874Z","published",null,[],"ai",[24,26,27,28],"distillation","reinforcement-learning","llm-research",[30],{"name":31,"url":32},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2609.29142",0,{"sections":35},[36,40,45,50,55,60,65,70,75,80,85,90,95,100],{"name":37,"slug":24,"count":38,"latest_published_at":39},"AI",4557,"2026-09-25T17:16:30.000Z",{"name":41,"slug":42,"count":43,"latest_published_at":44},"Security","security",741,"2026-09-25T15:52:13.000Z",{"name":46,"slug":47,"count":48,"latest_published_at":49},"Policy","policy",390,"2026-09-25T16:24:59.000Z",{"name":51,"slug":52,"count":53,"latest_published_at":54},"Deals","deals",256,"2026-09-25T17:00:53.000Z",{"name":56,"slug":57,"count":58,"latest_published_at":59},"Hardware","hardware",185,"2026-09-25T15:00:22.000Z",{"name":61,"slug":62,"count":63,"latest_published_at":64},"Science","science",140,"2026-09-25T11:55:23.000Z",{"name":66,"slug":67,"count":68,"latest_published_at":69},"Consumer Tech","consumer-tech",132,"2026-09-25T15:30:00.000Z",{"name":71,"slug":72,"count":73,"latest_published_at":74},"Software","software",88,"2026-09-24T23:06:55.000Z",{"name":76,"slug":77,"count":78,"latest_published_at":79},"Dev Tools","dev-tools",82,"2026-09-25T09:59:40.000Z",{"name":81,"slug":82,"count":83,"latest_published_at":84},"Startups","startups",76,"2026-09-25T18:33:59.000Z",{"name":86,"slug":87,"count":88,"latest_published_at":89},"Gaming","gaming",48,"2026-09-25T18:35:21.000Z",{"name":91,"slug":92,"count":93,"latest_published_at":94},"General","general",46,"2026-09-25T02:12:57.000Z",{"name":96,"slug":97,"count":98,"latest_published_at":99},"Reviews","reviews",30,"2026-09-24T20:07:31.000Z",{"name":101,"slug":102,"count":103,"latest_published_at":104},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]