[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-new-method-fixes-a-flaw-in-search-based-ai-math-training":10,"sections":34},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":29,"feedback":33,"feedback_at":22,"cost_usd":33,"total_tokens":33},9504,"new-method-fixes-a-flaw-in-search-based-ai-math-training","New Method Fixes a Flaw in Search-Based AI Math Training","A new training framework called APIVIS shows that adding search to AI math training only helps if you verify the search actually found something better.","A new paper shows that bolting search onto AI math training can backfire unless you check whether the search actually finds anything better.\n\nResearchers propose APIVIS, a training-time framework that folds a finite-budget Gumbel search into the chunk-by-chunk reasoning steps of a model learning math through reinforcement learning with verifiable rewards (RLVR). Each training rollout mixes a model's direct answer with a searched alternative, so when search turns up a better path, the difference becomes a usable reward signal instead of noise. The method also applies supervision only to the tokens search actually improved, which matters because the common GRPO training algorithm stalls when every answer in a group gets the same reward. The authors say this closes a specific gap in prior search-augmented RLVR work: adding search increases the variety of rollouts, but variety alone does not guarantee the resulting policy is actually better than what the model already does.\n\nThe team proves, rather than just claims, that selecting responses by value guidance raises the expected reward at each searched step, and that the guarantee holds even when the value estimate is imperfect. On standard math reasoning benchmarks across multiple model sizes, they report APIVIS beating other search-based RLVR methods by a clear margin.\n\nMath benchmarks are a favorite proving ground for reasoning papers precisely because gains are easy to measure and easy to overstate; whether chunk-level Gumbel search holds up on messier, real-world reasoning tasks is a question this paper does not answer.","[\"ai\",\"reinforcement-learning\",\"math-reasoning\",\"llms\"]","2026-10-02T04:00:00.000Z","2026-10-02T22:16:26.242Z","2026-10-02T22:16:32.700Z","published",null,[],"ai",[24,26,27,28],"reinforcement-learning","math-reasoning","llms",[30],{"name":31,"url":32},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2610.01080",0,{"sections":35},[36,39,43,47,52,57,61,66,71,76,81,86,91,96],{"name":37,"slug":24,"count":38,"latest_published_at":18},"AI",5859,{"name":40,"slug":41,"count":42,"latest_published_at":18},"Security","security",833,{"name":44,"slug":45,"count":46,"latest_published_at":18},"Policy","policy",438,{"name":48,"slug":49,"count":50,"latest_published_at":51},"Deals","deals",317,"2026-10-01T22:00:00.000Z",{"name":53,"slug":54,"count":55,"latest_published_at":56},"Hardware","hardware",198,"2026-10-01T17:38:48.000Z",{"name":58,"slug":59,"count":60,"latest_published_at":18},"Science","science",171,{"name":62,"slug":63,"count":64,"latest_published_at":65},"Consumer Tech","consumer-tech",155,"2026-10-01T19:54:10.000Z",{"name":67,"slug":68,"count":69,"latest_published_at":70},"Dev Tools","dev-tools",96,"2026-10-01T16:57:03.000Z",{"name":72,"slug":73,"count":74,"latest_published_at":75},"Software","software",93,"2026-09-30T21:41:11.000Z",{"name":77,"slug":78,"count":79,"latest_published_at":80},"Startups","startups",90,"2026-10-01T21:55:22.000Z",{"name":82,"slug":83,"count":84,"latest_published_at":85},"Gaming","gaming",53,"2026-10-02T02:50:39.000Z",{"name":87,"slug":88,"count":89,"latest_published_at":90},"General","general",50,"2026-09-30T21:37:54.000Z",{"name":92,"slug":93,"count":94,"latest_published_at":95},"Reviews","reviews",31,"2026-09-28T14:31:34.000Z",{"name":97,"slug":98,"count":99,"latest_published_at":100},"How-To","how-to",7,"2026-10-01T09:00:00.000Z"]