[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-study-traces-how-weight-decay-tames-giant-transformer-activations":10,"sections":34},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":29,"feedback":33,"feedback_at":22,"cost_usd":33,"total_tokens":33},9629,"study-traces-how-weight-decay-tames-giant-transformer-activations","Study Traces How Weight Decay Tames Giant Transformer Activations","Weight decay, a routine training knob, turns out to control how large a handful of rogue transformer activations grow before shrinking back.","A new study finds that an old regularization trick quietly decides how extreme a transformer's strangest neurons get.\n\nResearchers tracked the life cycle of massive activations, residual-stream values far bigger than normal activations that are linked to attention sinks. They found that which channels carry the sink varies across random seeds but locks in early in each training run. As training continues, neighboring channels fade out, and the sink concentrates onto a small set of redundant carriers. Through controlled training interventions, the team showed that weight decay drives the rise and fall: removing it near the peak lets the scale keep climbing, while keeping it causes a decline even at a constant learning rate.\n\nThis matters because gradient flow through the network tracks the collective size of these activations, not any single channel, which affects how models learn and how they hold up under pruning or quantization. The finding hands engineers a concrete lever, the decay coefficient, for shaping a model's internal behavior without touching its architecture.\n\nIt is a useful reminder that today's largest models still run on tuning knobs borrowed from far smaller networks decades ago.","[\"ai\",\"transformers\",\"machine-learning\",\"research\"]","2026-10-02T04:00:00.000Z","2026-10-03T03:48:56.572Z","2026-10-03T03:49:03.153Z","published",null,[],"ai",[24,26,27,28],"transformers","machine-learning","research",[30],{"name":31,"url":32},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2610.00423",0,{"sections":35},[36,39,43,47,52,56,60,65,70,75,80,85,90,95],{"name":37,"slug":24,"count":38,"latest_published_at":18},"AI",5976,{"name":40,"slug":41,"count":42,"latest_published_at":18},"Security","security",842,{"name":44,"slug":45,"count":46,"latest_published_at":18},"Policy","policy",438,{"name":48,"slug":49,"count":50,"latest_published_at":51},"Deals","deals",317,"2026-10-01T22:00:00.000Z",{"name":53,"slug":54,"count":55,"latest_published_at":18},"Hardware","hardware",199,{"name":57,"slug":58,"count":59,"latest_published_at":18},"Science","science",173,{"name":61,"slug":62,"count":63,"latest_published_at":64},"Consumer Tech","consumer-tech",155,"2026-10-01T19:54:10.000Z",{"name":66,"slug":67,"count":68,"latest_published_at":69},"Dev Tools","dev-tools",96,"2026-10-01T16:57:03.000Z",{"name":71,"slug":72,"count":73,"latest_published_at":74},"Software","software",93,"2026-09-30T21:41:11.000Z",{"name":76,"slug":77,"count":78,"latest_published_at":79},"Startups","startups",90,"2026-10-01T21:55:22.000Z",{"name":81,"slug":82,"count":83,"latest_published_at":84},"Gaming","gaming",53,"2026-10-02T02:50:39.000Z",{"name":86,"slug":87,"count":88,"latest_published_at":89},"General","general",50,"2026-09-30T21:37:54.000Z",{"name":91,"slug":92,"count":93,"latest_published_at":94},"Reviews","reviews",31,"2026-09-28T14:31:34.000Z",{"name":96,"slug":97,"count":98,"latest_published_at":99},"How-To","how-to",7,"2026-10-01T09:00:00.000Z"]