[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-not-all-hard-questions-need-more-thinking-study-finds":10,"sections":34},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":29,"feedback":33,"feedback_at":22,"cost_usd":33,"total_tokens":33},7717,"not-all-hard-questions-need-more-thinking-study-finds","Not All Hard Questions Need More Thinking, Study Finds","A new method trims AI reasoning length by 37 percent while boosting accuracy by targeting only the questions where longer thinking actually helps.","Researchers have found that reasoning models don't need to think longer on every hard question - just the ones sitting on the edge of solvable.\n\nReinforcement learning has gotten large language models better at math and coding by rewarding step-by-step reasoning, but it has a side effect: models ramble on simple questions and cut off too early on hard ones. Most existing fixes assume harder questions always benefit from more tokens, so they hand out bigger reasoning budgets as difficulty rises. A team publishing on arXiv challenges that assumption, showing the real payoff from extra reasoning length is concentrated on questions that are only partially solvable, not the hardest ones overall. They also found that explicit rewards for longer or shorter answers can distort training in unintended ways. Their fix, called CARE (Contrastive Accuracy Reward Estimation), compares sampled responses per question during training and adjusts length rewards accordingly, inside the standard Group Relative Policy Optimization setup, without adding hyperparameters or extra inference cost.\n\nThis challenges a design assumption baked into a lot of recent reasoning-model tooling: that difficulty and reasoning length should scale together. If length only helps on a narrow band of near-miss questions, budget-by-difficulty schemes are solving the wrong variable, burning compute on both easy and very hard prompts alike. For anyone paying per token for inference-heavy reasoning models, that is a real cost lever, not an academic footnote.\n\nThe reported numbers - up to 4% higher Pass@1 with 37% shorter reasoning chains - sound tidy for one paper; the real test is whether it holds once other labs try it on models it wasn't tuned for.","[\"ai\",\"reasoning-models\",\"reinforcement-learning\",\"research\"]","2026-09-25T04:00:00.000Z","2026-09-25T06:10:37.468Z","2026-09-25T06:10:43.042Z","published",null,[],"ai",[24,26,27,28],"reasoning-models","reinforcement-learning","research",[30],{"name":31,"url":32},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2609.29664",0,{"sections":35},[36,39,44,49,54,59,64,69,74,79,84,89,93,98],{"name":37,"slug":24,"count":38,"latest_published_at":18},"AI",4466,{"name":40,"slug":41,"count":42,"latest_published_at":43},"Security","security",729,"2026-09-24T19:54:21.000Z",{"name":45,"slug":46,"count":47,"latest_published_at":48},"Policy","policy",386,"2026-09-24T23:50:55.000Z",{"name":50,"slug":51,"count":52,"latest_published_at":53},"Deals","deals",237,"2026-09-24T22:00:00.000Z",{"name":55,"slug":56,"count":57,"latest_published_at":58},"Hardware","hardware",182,"2026-09-25T01:25:53.000Z",{"name":60,"slug":61,"count":62,"latest_published_at":63},"Science","science",138,"2026-09-24T18:24:52.000Z",{"name":65,"slug":66,"count":67,"latest_published_at":68},"Consumer Tech","consumer-tech",128,"2026-09-24T19:24:34.000Z",{"name":70,"slug":71,"count":72,"latest_published_at":73},"Software","software",88,"2026-09-24T23:06:55.000Z",{"name":75,"slug":76,"count":77,"latest_published_at":78},"Dev Tools","dev-tools",79,"2026-09-22T22:21:13.000Z",{"name":80,"slug":81,"count":82,"latest_published_at":83},"Startups","startups",71,"2026-09-24T20:45:00.000Z",{"name":85,"slug":86,"count":87,"latest_published_at":88},"Gaming","gaming",46,"2026-09-24T17:52:29.000Z",{"name":90,"slug":91,"count":87,"latest_published_at":92},"General","general","2026-09-25T02:12:57.000Z",{"name":94,"slug":95,"count":96,"latest_published_at":97},"Reviews","reviews",30,"2026-09-24T20:07:31.000Z",{"name":99,"slug":100,"count":101,"latest_published_at":102},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]