[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-new-rl-method-times-policy-updates-by-information-not-tokens":10,"sections":34},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":29,"feedback":33,"feedback_at":22,"cost_usd":33,"total_tokens":33},7368,"new-rl-method-times-policy-updates-by-information-not-tokens","New RL Method Times Policy Updates by Information, Not Tokens","A new technique called InfoPPO paces reinforcement learning by information density instead of token count, fixing instability in long AI reasoning chains.","Researchers have found a subtle bug in how AI reasoning models learn from feedback: the training math was measuring time by counting tokens, not by tracking where the actual information showed up.\n\nA new paper introduces InfoPPO, a twist on the reinforcement learning method (PPO) used to train large language models on verifiable-reward tasks like math problems, where standard training treats each generated token as one uniform tick of the clock even though real reasoning is lumpy: some tokens carry far more new information than others. InfoPPO instead paces training by information density, letting it apply discounting (a way of deciding how much credit earlier steps get for a correct final answer) without either losing long-range structure or drowning out the final reward, and it adjusts how much the policy can change at each token based on that token's information density. Tested on Qwen3 models across five competition-style math benchmarks, InfoPPO beat standard PPO baselines and stayed stable at discount settings that made ordinary token-time PPO fall apart.\n\nThe fix targets a real bottleneck: as reasoning chains get longer, RL training either loses track of which steps mattered or lets the final answer's reward signal get diluted across thousands of tokens. That tension has capped how well RL scales to genuinely long chains of thought, and InfoPPO's information-time reframing is a more principled fix than the length penalties and heuristic reward shaping teams have leaned on so far.\n\nIt is only tested on math benchmarks with clean, verifiable rewards, though, so it is worth watching whether the same trick survives contact with messier, harder-to-verify agent tasks.","[\"reinforcement-learning\",\"llm-reasoning\",\"ai-research\"]","2026-09-23T04:00:00.000Z","2026-09-23T09:53:58.550Z","2026-09-23T09:54:04.009Z","published",null,[],"ai",[26,27,28],"reinforcement-learning","llm-reasoning","ai-research",[30],{"name":31,"url":32},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2609.24380",0,{"sections":35},[36,39,43,47,52,56,61,66,71,76,81,86,91,96],{"name":37,"slug":24,"count":38,"latest_published_at":18},"AI",4344,{"name":40,"slug":41,"count":42,"latest_published_at":18},"Security","security",713,{"name":44,"slug":45,"count":46,"latest_published_at":18},"Policy","policy",370,{"name":48,"slug":49,"count":50,"latest_published_at":51},"Deals","deals",206,"2026-09-23T09:43:46.000Z",{"name":53,"slug":54,"count":55,"latest_published_at":18},"Hardware","hardware",169,{"name":57,"slug":58,"count":59,"latest_published_at":60},"Science","science",134,"2026-09-23T09:00:00.000Z",{"name":62,"slug":63,"count":64,"latest_published_at":65},"Consumer Tech","consumer-tech",110,"2026-09-22T20:00:00.000Z",{"name":67,"slug":68,"count":69,"latest_published_at":70},"Software","software",81,"2026-09-23T09:56:13.000Z",{"name":72,"slug":73,"count":74,"latest_published_at":75},"Dev Tools","dev-tools",79,"2026-09-22T22:21:13.000Z",{"name":77,"slug":78,"count":79,"latest_published_at":80},"Startups","startups",65,"2026-09-22T22:06:48.000Z",{"name":82,"slug":83,"count":84,"latest_published_at":85},"Gaming","gaming",45,"2026-09-22T15:35:06.000Z",{"name":87,"slug":88,"count":89,"latest_published_at":90},"General","general",43,"2026-09-21T23:48:56.000Z",{"name":92,"slug":93,"count":94,"latest_published_at":95},"Reviews","reviews",27,"2026-09-22T13:00:00.000Z",{"name":97,"slug":98,"count":99,"latest_published_at":100},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]