[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-a-fine-tuning-shortcut-skips-backpropagation-triples-throughput":10,"sections":34},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":29,"feedback":33,"feedback_at":22,"cost_usd":33,"total_tokens":33},5276,"a-fine-tuning-shortcut-skips-backpropagation-triples-throughput","A Fine-Tuning Shortcut Skips Backpropagation, Triples Throughput","A new fine-tuning method estimates gradients at the output layer instead of backpropagating through the model, cutting cost without hurting performance.","A new training trick lets researchers fine-tune large language models without backpropagating through most of the network - and it roughly triples throughput while using less memory.\n\nThe method, called Forward-Pass-Only MLP training (FPO), starts from an observation: in the late layers of a transformer, the error at the model's output layer already approximates the true gradient. Across six public models, the two turned out to be correlated with a cosine similarity of 0.47 to 0.59 - close enough to be useful, not close enough to be exact. The researchers built a two-minute diagnostic that checks, layer by layer, where this approximation holds. FPO then takes a single error signal from the output and applies it directly to those layers, skipping the backward pass and the computational graph that normal backpropagation needs. Tested on OLMo-2-7B, Qwen3-8B, and Falcon3-7B, it delivered 2.7x to 3.2x the throughput of standard fine-tuning and cut peak memory by about 40 percent, while improving in-domain perplexity.\n\nThe more interesting number isn't the speedup - it's what stayed the same. On MMLU, ARC-Challenge, HellaSwag, and Winogrande, FPO-tuned models scored within seed-noise of the untouched baseline. Standard full-network fine-tuning, the paper notes, does not reliably manage that. For anyone building specialized versions of open models, that's the harder problem: making a model better at one thing without quietly making it worse at everything else.\n\nWorth noting: this is one arXiv preprint, not yet peer-reviewed, and the underlying approximation is a correlation in the high 0.4s and 0.5s, not a substitute for the real gradient. A localized version of ordinary fine-tuning gets similar results but costs 2.2 times as much in wall-clock time - so the savings here come specifically from ditching the backward pass, not just training fewer layers.","[\"ai\",\"fine-tuning\",\"machine-learning\",\"research\"]","2026-08-18T04:00:00.000Z","2026-08-18T12:58:54.776Z","2026-08-18T12:59:06.555Z","published",null,[],"ai",[24,26,27,28],"fine-tuning","machine-learning","research",[30],{"name":31,"url":32},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2608.14563",0,{"sections":35},[36,40,44,49,54,59,64,69,74,78,83,88,93,98],{"name":37,"slug":24,"count":38,"latest_published_at":39},"AI",3293,"2026-08-20T04:00:00.000Z",{"name":41,"slug":42,"count":43,"latest_published_at":39},"Security","security",435,{"name":45,"slug":46,"count":47,"latest_published_at":48},"Policy","policy",210,"2026-08-19T09:32:27.000Z",{"name":50,"slug":51,"count":52,"latest_published_at":53},"Deals","deals",179,"2026-06-29T20:02:07.000Z",{"name":55,"slug":56,"count":57,"latest_published_at":58},"Hardware","hardware",140,"2026-08-19T18:25:42.000Z",{"name":60,"slug":61,"count":62,"latest_published_at":63},"Consumer Tech","consumer-tech",95,"2026-08-18T16:05:00.000Z",{"name":65,"slug":66,"count":67,"latest_published_at":68},"Science","science",90,"2026-08-19T18:41:02.000Z",{"name":70,"slug":71,"count":72,"latest_published_at":73},"Software","software",73,"2026-08-18T07:51:50.000Z",{"name":75,"slug":76,"count":77,"latest_published_at":18},"Dev Tools","dev-tools",69,{"name":79,"slug":80,"count":81,"latest_published_at":82},"Startups","startups",47,"2026-08-19T19:13:46.000Z",{"name":84,"slug":85,"count":86,"latest_published_at":87},"Gaming","gaming",41,"2026-07-09T04:00:00.000Z",{"name":89,"slug":90,"count":91,"latest_published_at":92},"General","general",33,"2026-08-18T22:18:13.000Z",{"name":94,"slug":95,"count":96,"latest_published_at":97},"Reviews","reviews",20,"2026-06-24T12:00:01.000Z",{"name":99,"slug":100,"count":101,"latest_published_at":102},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]