[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-why-ai-benchmarks-overrate-compressed-models":10,"sections":34},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":29,"feedback":33,"feedback_at":22,"cost_usd":33,"total_tokens":33},6970,"why-ai-benchmarks-overrate-compressed-models","Why AI Benchmarks Overrate Compressed Models","A distilled 7B model matched its teacher on paper but fell 20.8 points behind once it had to actually generate answers, researchers found.","A widely used way of grading compressed AI models has been quietly lying to researchers.\n\nA new paper from researchers building GenDistill, a pipeline for distilling large Transformers into more efficient hybrid models, found that standard benchmark scoring badly overstates how good these shrunk models actually are. Most distillation studies grade multiple-choice benchmarks by ranking candidate answers with log-likelihood scores instead of making the model generate an answer on its own. The researchers tested a 7B distilled model both ways: under log-likelihood scoring it scored within 0.2 percentage points of its full-size teacher model, but when forced to generate answers autoregressively, it fell 20.8 points behind. Using Qwen3-0.6B as a testbed, they ran the same double-check across six design choices, including training data selection and which layers get frozen during training, and found the log-likelihood method consistently made student models look better than they performed in practice.\n\nThis matters because efficient hybrid models are supposed to make AI cheaper to run, and the researchers' best recipe cut KV cache memory by up to 75% and sped up time-to-first-token by 2-4x at 128K-token contexts, while keeping 86-90% of teacher accuracy. But if the standard grading method hides how much quality a model actually loses, teams choosing between distillation recipes could be picking worse models without realizing it.\n\nPerplexity scores look tidy on a leaderboard, but asking a model to actually write an answer is where the shortcuts show.","[\"ai\",\"llms\",\"benchmarks\",\"model-distillation\"]","2026-09-18T04:00:00.000Z","2026-09-19T01:04:32.727Z","2026-09-19T01:04:44.630Z","published",null,[],"ai",[24,26,27,28],"llms","benchmarks","model-distillation",[30],{"name":31,"url":32},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2603.26556",0,{"sections":35},[36,39,43,48,53,57,61,66,70,75,80,85,90,95],{"name":37,"slug":24,"count":38,"latest_published_at":18},"AI",4082,{"name":40,"slug":41,"count":42,"latest_published_at":18},"Security","security",661,{"name":44,"slug":45,"count":46,"latest_published_at":47},"Policy","policy",339,"2026-09-17T12:00:00.000Z",{"name":49,"slug":50,"count":51,"latest_published_at":52},"Deals","deals",179,"2026-06-29T20:02:07.000Z",{"name":54,"slug":55,"count":56,"latest_published_at":18},"Hardware","hardware",155,{"name":58,"slug":59,"count":60,"latest_published_at":18},"Science","science",125,{"name":62,"slug":63,"count":64,"latest_published_at":65},"Consumer Tech","consumer-tech",99,"2026-09-09T17:27:33.000Z",{"name":67,"slug":68,"count":69,"latest_published_at":18},"Dev Tools","dev-tools",78,{"name":71,"slug":72,"count":73,"latest_published_at":74},"Software","software",75,"2026-09-10T20:41:21.000Z",{"name":76,"slug":77,"count":78,"latest_published_at":79},"Startups","startups",55,"2026-09-09T23:14:29.000Z",{"name":81,"slug":82,"count":83,"latest_published_at":84},"Gaming","gaming",43,"2026-09-10T12:18:06.000Z",{"name":86,"slug":87,"count":88,"latest_published_at":89},"General","general",41,"2026-09-08T01:57:23.000Z",{"name":91,"slug":92,"count":93,"latest_published_at":94},"Reviews","reviews",20,"2026-06-24T12:00:01.000Z",{"name":96,"slug":97,"count":98,"latest_published_at":99},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]