[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-researchers-teach-ai-models-to-learn-from-flawed-teachers":10,"sections":34},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":29,"feedback":33,"feedback_at":22,"cost_usd":33,"total_tokens":33},6613,"researchers-teach-ai-models-to-learn-from-flawed-teachers","Researchers Teach AI Models to Learn From Flawed Teachers","A new technique salvages useful signal from failed AI teacher demonstrations, cutting training compute roughly in half compared to prior methods.","A new training method squeezes useful lessons out of AI teacher models even when those teachers get the answer wrong.\n\nThe technique targets offline on-policy distillation, a way of shrinking large AI models into smaller, cheaper ones by having a \"student\" model learn from a \"teacher\" model's example trajectories, collected once and reused throughout training. The catch: when a teacher fails a problem, that failure was assumed to poison the whole trajectory, so researchers built a workaround. Their method trains on the problems the teacher solved correctly, then checks how much that success-only training shifts the likelihood of each token appearing in the failed trajectories - a signal for which parts of a bad trajectory are still worth learning from. Tested on math reasoning and code generation, it improved on a standard offline baseline by up to 2.7 percentage points and matched or beat online distillation methods that re-run the teacher live.\n\nThe efficiency angle is the real story. Online distillation methods needed 3 GPUs and 36 to 48 GPU hours in the paper's comparisons; this offline approach hit comparable or better results with 2 GPUs and about 22 GPU hours, without generating any new teacher data. For teams distilling frontier models into smaller deployable ones, that is a meaningful compute discount, not a rounding error.\n\nIt is one arXiv preprint, not a peer-reviewed result, and the gains are benchmark-specific to math and code tasks. Whether the trick holds up on messier, more open-ended teacher outputs is the next question worth asking.","[\"ai\",\"machine-learning\",\"model-training\",\"llm-research\"]","2026-09-17T04:00:00.000Z","2026-09-18T02:38:29.852Z","2026-09-18T02:38:41.836Z","published",null,[],"ai",[24,26,27,28],"machine-learning","model-training","llm-research",[30],{"name":31,"url":32},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2609.18321",0,{"sections":35},[36,40,44,49,54,58,62,67,72,76,81,86,91,96],{"name":37,"slug":24,"count":38,"latest_published_at":39},"AI",3853,"2026-09-17T08:27:09.000Z",{"name":41,"slug":42,"count":43,"latest_published_at":18},"Security","security",648,{"name":45,"slug":46,"count":47,"latest_published_at":48},"Policy","policy",338,"2026-09-11T04:00:00.000Z",{"name":50,"slug":51,"count":52,"latest_published_at":53},"Deals","deals",179,"2026-06-29T20:02:07.000Z",{"name":55,"slug":56,"count":57,"latest_published_at":18},"Hardware","hardware",154,{"name":59,"slug":60,"count":61,"latest_published_at":18},"Science","science",114,{"name":63,"slug":64,"count":65,"latest_published_at":66},"Consumer Tech","consumer-tech",99,"2026-09-09T17:27:33.000Z",{"name":68,"slug":69,"count":70,"latest_published_at":71},"Software","software",75,"2026-09-10T20:41:21.000Z",{"name":73,"slug":74,"count":75,"latest_published_at":18},"Dev Tools","dev-tools",73,{"name":77,"slug":78,"count":79,"latest_published_at":80},"Startups","startups",55,"2026-09-09T23:14:29.000Z",{"name":82,"slug":83,"count":84,"latest_published_at":85},"Gaming","gaming",43,"2026-09-10T12:18:06.000Z",{"name":87,"slug":88,"count":89,"latest_published_at":90},"General","general",41,"2026-09-08T01:57:23.000Z",{"name":92,"slug":93,"count":94,"latest_published_at":95},"Reviews","reviews",20,"2026-06-24T12:00:01.000Z",{"name":97,"slug":98,"count":99,"latest_published_at":100},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]