[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-new-theory-explains-why-wobbly-training-helps-ai-models-learn":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},9317,"new-theory-explains-why-wobbly-training-helps-ai-models-learn","New Theory Explains Why Wobbly Training Helps AI Models Learn","Researchers built a rigorous math model explaining why training with large, unstable learning rates can make neural networks generalize better.","A team of mathematicians has worked out a rigorous theory for one of deep learning's stranger habits: training that bounces instead of smoothly descending.\n\nWhen a neural network trains with a learning rate large enough, its loss doesn't fall smoothly. It oscillates, a pattern known as the edge of stability, and researchers have long noticed this jittery process often produces models that generalize better than careful, slow training. A new paper introduces mean-fluctuation dynamics, a continuous-time model that tracks both the averaged training path and the size of its oscillations as linked variables. The authors derive the model rigorously from gradient descent in a simplified sharp-valley setup, map its stable resting points, and extend the analysis to wide two-layer neural networks, proving the math holds up and identifying conditions under which training provably converges.\n\nEdge-of-stability training already happens by default across much of deep learning. It's usually a side effect of picking a learning rate large enough to train fast, not a deliberate strategy. A provable model for why that wobbly process can still converge, and converge well, gives researchers a mathematical foothold for learning-rate choices that have mostly been justified by trial and error and GPU budgets.\n\nThe strongest guarantees here still live in two-layer-network territory, so whether this framework scales to the sprawling architectures behind today's large language models remains open work.","[\"edge-of-stability\",\"gradient-descent\",\"deep-learning-theory\",\"neural-networks\"]","2026-10-01T04:00:00.000Z","2026-10-02T09:04:08.930Z","2026-10-02T09:04:11.433Z","published",null,[],"ai",[26,27,28,29],"edge-of-stability","gradient-descent","deep-learning-theory","neural-networks",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2606.05326",0,{"sections":36},[37,41,45,50,55,60,64,69,74,78,83,88,93,98],{"name":38,"slug":24,"count":39,"latest_published_at":40},"AI",5693,"2026-10-01T12:05:27.000Z",{"name":42,"slug":43,"count":44,"latest_published_at":18},"Security","security",820,{"name":46,"slug":47,"count":48,"latest_published_at":49},"Policy","policy",431,"2026-10-01T11:08:42.000Z",{"name":51,"slug":52,"count":53,"latest_published_at":54},"Deals","deals",308,"2026-10-01T12:30:00.000Z",{"name":56,"slug":57,"count":58,"latest_published_at":59},"Hardware","hardware",197,"2026-10-01T11:37:06.000Z",{"name":61,"slug":62,"count":63,"latest_published_at":18},"Science","science",165,{"name":65,"slug":66,"count":67,"latest_published_at":68},"Consumer Tech","consumer-tech",150,"2026-10-01T11:59:27.000Z",{"name":70,"slug":71,"count":72,"latest_published_at":73},"Dev Tools","dev-tools",93,"2026-10-01T02:30:48.000Z",{"name":75,"slug":76,"count":72,"latest_published_at":77},"Software","software","2026-09-30T21:41:11.000Z",{"name":79,"slug":80,"count":81,"latest_published_at":82},"Startups","startups",84,"2026-09-30T20:39:09.000Z",{"name":84,"slug":85,"count":86,"latest_published_at":87},"Gaming","gaming",51,"2026-09-30T16:24:30.000Z",{"name":89,"slug":90,"count":91,"latest_published_at":92},"General","general",50,"2026-09-30T21:37:54.000Z",{"name":94,"slug":95,"count":96,"latest_published_at":97},"Reviews","reviews",31,"2026-09-28T14:31:34.000Z",{"name":99,"slug":100,"count":101,"latest_published_at":102},"How-To","how-to",7,"2026-10-01T09:00:00.000Z"]