[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-a-multiple-choice-trick-speeds-up-small-ai-agents":10,"sections":34},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":29,"feedback":33,"feedback_at":22,"cost_usd":33,"total_tokens":33},9767,"a-multiple-choice-trick-speeds-up-small-ai-agents","A Multiple Choice Trick Speeds Up Small AI Agents","A new framework lets small multimodal AI agents pick from preset reasoning options instead of writing it out, cutting inference time by up to half.","A team of researchers just made small AI agents stop rambling before every move.\n\nThe new framework, called Selection-based Structured Reasoning (SSR), swaps free-form chain-of-thought for multiple choice. Instead of generating a fresh paragraph of reasoning at each step, the model picks from a fixed menu of pre-written reasoning candidates, scored by likelihood against the current context. That scoring runs in parallel using a shared cache, rather than one slow token-by-token generation pass. The team tested SSR on 2B and 4B parameter models across seven multimodal search benchmarks, using both reinforcement learning and supervised fine-tuning.\n\nFor small models, that rambling reasoning was mostly theater: it cost compute without improving the answer. SSR cut per-turn reasoning latency by over 90% and total inference latency by 28-54%, while matching the success rate of comparable search agents at the same scale. That is the kind of efficiency gain that matters for running agents on limited hardware, not a marginal benchmark bump.\n\nIt will not make a model think better - it just makes thinking cheaper, which for most real-world deployments is the harder problem anyway.","[\"ai\",\"ai-agents\",\"efficiency\",\"research\"]","2026-10-02T04:00:00.000Z","2026-10-03T09:40:22.788Z","2026-10-03T09:40:27.383Z","published",null,[],"ai",[24,26,27,28],"ai-agents","efficiency","research",[30],{"name":31,"url":32},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2610.01892",0,{"sections":35},[36,39,43,47,52,56,60,65,70,75,80,85,90,95],{"name":37,"slug":24,"count":38,"latest_published_at":18},"AI",6050,{"name":40,"slug":41,"count":42,"latest_published_at":18},"Security","security",848,{"name":44,"slug":45,"count":46,"latest_published_at":18},"Policy","policy",439,{"name":48,"slug":49,"count":50,"latest_published_at":51},"Deals","deals",317,"2026-10-01T22:00:00.000Z",{"name":53,"slug":54,"count":55,"latest_published_at":18},"Hardware","hardware",199,{"name":57,"slug":58,"count":59,"latest_published_at":18},"Science","science",176,{"name":61,"slug":62,"count":63,"latest_published_at":64},"Consumer Tech","consumer-tech",155,"2026-10-01T19:54:10.000Z",{"name":66,"slug":67,"count":68,"latest_published_at":69},"Dev Tools","dev-tools",96,"2026-10-01T16:57:03.000Z",{"name":71,"slug":72,"count":73,"latest_published_at":74},"Software","software",93,"2026-09-30T21:41:11.000Z",{"name":76,"slug":77,"count":78,"latest_published_at":79},"Startups","startups",90,"2026-10-01T21:55:22.000Z",{"name":81,"slug":82,"count":83,"latest_published_at":84},"Gaming","gaming",53,"2026-10-02T02:50:39.000Z",{"name":86,"slug":87,"count":88,"latest_published_at":89},"General","general",50,"2026-09-30T21:37:54.000Z",{"name":91,"slug":92,"count":93,"latest_published_at":94},"Reviews","reviews",31,"2026-09-28T14:31:34.000Z",{"name":96,"slug":97,"count":98,"latest_published_at":99},"How-To","how-to",7,"2026-10-01T09:00:00.000Z"]