[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-new-method-trains-ai-reasoning-models-without-sharing-raw-data":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},10877,"new-method-trains-ai-reasoning-models-without-sharing-raw-data","New Method Trains AI Reasoning Models Without Sharing Raw Data","Fed-GRPO fine-tunes AI reasoning models across separate, siloed datasets using reinforcement learning, slashing the data transfer needed by up to 621 times.","A new training method lets AI labs fine-tune reasoning models together without ever pooling their data.\n\nThe approach, called Fed-GRPO, adapts Group Relative Policy Optimization, the reinforcement-learning technique behind many of today's reasoning-focused language models, for federated learning, where multiple parties train jointly but keep their raw data local. Instead of adding extra computation to manage this, Fed-GRPO reuses the reward statistics that GRPO already produces during training as a free signal. That signal drives three mechanisms: weighting each participant's updates by how much learning signal it is producing, recalibrating each participant's training objective based on the gap between its local performance and the group's, and allocating communication bandwidth to whichever updates carry the most information. On math reasoning benchmarks, Fed-GRPO beat a standard federated-averaging baseline and came close to matching a fully centralized training run, while cutting communication by 32 times with no loss in accuracy, and by up to 621 times when bandwidth is tightly constrained, at the cost of some accuracy.\n\nThat matters because privacy rules and competitive concerns keep a lot of valuable training data locked inside separate organizations, hospitals, banks, and rival companies that cannot legally or practically share it. Until now, groups like that faced a tradeoff: use weaker federated techniques, or skip reinforcement-learning fine-tuning for reasoning tasks entirely. Fed-GRPO is a specific attempt to close that gap using only signals the training process already generates, rather than new infrastructure.\n\nThe catch is that \"approaches centralized performance\" is not the same as matching it, and so far the results are limited to math reasoning benchmarks, which are unusually clean compared to the messier reasoning tasks most businesses actually care about.","[\"federated-learning\",\"reinforcement-learning\",\"llm-reasoning\",\"ai-research\"]","2026-10-09T04:00:00.000Z","2026-10-09T19:36:31.446Z","2026-10-09T19:36:34.802Z","published",null,[],"ai",[26,27,28,29],"federated-learning","reinforcement-learning","llm-reasoning","ai-research",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2610.11502",0,{"sections":36},[37,40,44,49,54,59,63,68,73,78,83,88,93,98],{"name":38,"slug":24,"count":39,"latest_published_at":18},"AI",6619,{"name":41,"slug":42,"count":43,"latest_published_at":18},"Security","security",926,{"name":45,"slug":46,"count":47,"latest_published_at":48},"Policy","policy",486,"2026-10-08T22:40:11.000Z",{"name":50,"slug":51,"count":52,"latest_published_at":53},"Deals","deals",474,"2026-10-08T22:00:00.000Z",{"name":55,"slug":56,"count":57,"latest_published_at":58},"Hardware","hardware",229,"2026-10-08T20:47:10.000Z",{"name":60,"slug":61,"count":62,"latest_published_at":18},"Science","science",192,{"name":64,"slug":65,"count":66,"latest_published_at":67},"Consumer Tech","consumer-tech",181,"2026-10-08T23:26:35.000Z",{"name":69,"slug":70,"count":71,"latest_published_at":72},"Startups","startups",117,"2026-10-08T16:45:00.000Z",{"name":74,"slug":75,"count":76,"latest_published_at":77},"Software","software",114,"2026-10-08T17:57:01.000Z",{"name":79,"slug":80,"count":81,"latest_published_at":82},"Dev Tools","dev-tools",105,"2026-10-07T16:59:11.000Z",{"name":84,"slug":85,"count":86,"latest_published_at":87},"General","general",66,"2026-10-09T04:46:11.000Z",{"name":89,"slug":90,"count":91,"latest_published_at":92},"Gaming","gaming",58,"2026-10-08T20:08:45.000Z",{"name":94,"slug":95,"count":96,"latest_published_at":97},"Reviews","reviews",34,"2026-10-08T14:00:22.000Z",{"name":99,"slug":100,"count":101,"latest_published_at":102},"How-To","how-to",8,"2026-10-05T09:00:00.000Z"]