[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-new-method-tunes-how-ai-models-learn-from-their-own-answers":10,"sections":34},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":29,"feedback":33,"feedback_at":22,"cost_usd":33,"total_tokens":33},10904,"new-method-tunes-how-ai-models-learn-from-their-own-answers","New Method Tunes How AI Models Learn From Their Own Answers","A new technique teaches small AI models which of their own words are worth learning from, lifting math-reasoning accuracy over standard token weighting.","A new AI training technique teaches models which of their own words are actually worth learning from - and it works better than treating every word the same.\n\nThe approach, called MetaOPD, improves on-policy distillation, a method where a smaller student model generates its own answers and a larger teacher model scores each word in that answer to guide further training. Normally, every word gets graded with equal weight, even filler words that teach the student nothing. MetaOPD adds a small second network that learns, through repeated trial and error, which words actually improve the student's performance - instead of relying on a fixed, human-designed formula. Researchers tested it on math-reasoning problems using two small student models, 0.6 billion and 1.7 billion parameters, comparing it against seven existing weighting methods across six math benchmarks and three unrelated test sets.\n\nThe results matter because most companies run small, cheap models, not frontier-scale ones, so getting more out of a 1-2 billion-parameter model without buying more compute or data is the more practical lever. The gains were real but not dramatic: roughly 2 percentage points better average accuracy, and close to 6 points better on pass-rate across repeated attempts, for both model sizes tested.\n\nStill, this is one paper's math benchmarks, not a deployed product - call it a promising tweak to a training recipe, not a reason to retire your existing fine-tuning pipeline just yet.","[\"ai\",\"model training\",\"on-policy distillation\",\"small language models\"]","2026-10-09T04:00:00.000Z","2026-10-09T20:52:51.153Z","2026-10-09T20:52:56.665Z","published",null,[],"ai",[24,26,27,28],"model training","on-policy distillation","small language models",[30],{"name":31,"url":32},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2610.11989",0,{"sections":35},[36,39,43,48,53,57,61,66,71,76,81,86,91,96],{"name":37,"slug":24,"count":38,"latest_published_at":18},"AI",6690,{"name":40,"slug":41,"count":42,"latest_published_at":18},"Security","security",930,{"name":44,"slug":45,"count":46,"latest_published_at":47},"Policy","policy",486,"2026-10-08T22:40:11.000Z",{"name":49,"slug":50,"count":51,"latest_published_at":52},"Deals","deals",474,"2026-10-08T22:00:00.000Z",{"name":54,"slug":55,"count":56,"latest_published_at":18},"Hardware","hardware",231,{"name":58,"slug":59,"count":60,"latest_published_at":18},"Science","science",192,{"name":62,"slug":63,"count":64,"latest_published_at":65},"Consumer Tech","consumer-tech",181,"2026-10-08T23:26:35.000Z",{"name":67,"slug":68,"count":69,"latest_published_at":70},"Startups","startups",117,"2026-10-08T16:45:00.000Z",{"name":72,"slug":73,"count":74,"latest_published_at":75},"Software","software",114,"2026-10-08T17:57:01.000Z",{"name":77,"slug":78,"count":79,"latest_published_at":80},"Dev Tools","dev-tools",105,"2026-10-07T16:59:11.000Z",{"name":82,"slug":83,"count":84,"latest_published_at":85},"General","general",66,"2026-10-09T04:46:11.000Z",{"name":87,"slug":88,"count":89,"latest_published_at":90},"Gaming","gaming",58,"2026-10-08T20:08:45.000Z",{"name":92,"slug":93,"count":94,"latest_published_at":95},"Reviews","reviews",34,"2026-10-08T14:00:22.000Z",{"name":97,"slug":98,"count":99,"latest_published_at":100},"How-To","how-to",8,"2026-10-05T09:00:00.000Z"]