[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-new-benchmark-finds-tool-design-matters-more-than-model-size":10,"sections":46},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":35,"tags":36,"sources":41,"feedback":45,"feedback_at":22,"cost_usd":45,"total_tokens":45},7361,"new-benchmark-finds-tool-design-matters-more-than-model-size","New Benchmark Finds Tool Design Matters More Than Model Size","A new open-source benchmark shows that how you split up an AI agent's toolbox matters more than how big the underlying model is.","A new benchmark says the way you chop up an AI agent's toolbox matters more than which model you plug into it.\n\nResearchers built MCP-GRANITE, an open-source benchmark for testing tool-interface granularity in agents that use the Model Context Protocol to call outside tools. It runs 81 multi-step tasks across 9 domains, each instantiated at four granularity levels, from many narrow single-purpose tools down to one do-everything tool. Nine locally deployed models, from 268 million to 20.9 billion parameters, were tested across 8,748 trials - three runs per model-scenario-granularity combination, which is how 9 models times 81 scenarios times 4 granularity levels (2,916 configurations) becomes 8,748 total trials. The team measured task completion, tool-selection accuracy, argument accuracy, latency, and resource use.\n\nA four-tool setup won. It beat the finest-grained split by 16.4% on task completion and the single monolithic tool by 33.6%, while nearly doubling argument accuracy. Model size barely predicted task completion and mostly just predicted latency - a 3.2B-parameter model at the right granularity beat a 20.9B model at the wrong one.\n\nFor anyone shipping agents on constrained hardware, that's the more useful lever: before reaching for a bigger model, check whether the tool list is the actual bottleneck.","[\"ai agents\",\"mcp\",\"edge ai\",\"benchmarking\"]","2026-09-23T04:00:00.000Z","2026-09-23T09:35:40.760Z","2026-09-23T09:35:45.885Z","published",null,[24,30],{"id":25,"reviewer":26,"round":27,"reason":28,"status":29},"publisher-r1","publisher",1,"The trial count doesn't check out: 9 models × 81 scenarios × 4 granularity levels = 2,916, not the 8,748 stated in the body.","resolved",{"id":31,"reviewer":32,"round":33,"reason":34,"status":29},"editor-r2","editor",2,"The trial count still doesn't reconcile: 9 models × 81 scenarios × 4 granularity levels = 2,916, not the 8,748 cited — either correct the math or explain the multiplier (e.g., repeated runs per configuration) before publishing.","ai",[37,38,39,40],"ai agents","mcp","edge ai","benchmarking",[42],{"name":43,"url":44},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2609.24161",0,{"sections":47},[48,51,55,60,65,69,73,78,83,88,93,98,103,108],{"name":49,"slug":35,"count":50,"latest_published_at":18},"AI",4314,{"name":52,"slug":53,"count":54,"latest_published_at":18},"Security","security",710,{"name":56,"slug":57,"count":58,"latest_published_at":59},"Policy","policy",369,"2026-09-23T02:13:52.000Z",{"name":61,"slug":62,"count":63,"latest_published_at":64},"Deals","deals",202,"2026-09-22T23:00:04.000Z",{"name":66,"slug":67,"count":68,"latest_published_at":18},"Hardware","hardware",169,{"name":70,"slug":71,"count":72,"latest_published_at":18},"Science","science",133,{"name":74,"slug":75,"count":76,"latest_published_at":77},"Consumer Tech","consumer-tech",110,"2026-09-22T20:00:00.000Z",{"name":79,"slug":80,"count":81,"latest_published_at":82},"Software","software",80,"2026-09-22T23:32:52.000Z",{"name":84,"slug":85,"count":86,"latest_published_at":87},"Dev Tools","dev-tools",79,"2026-09-22T22:21:13.000Z",{"name":89,"slug":90,"count":91,"latest_published_at":92},"Startups","startups",65,"2026-09-22T22:06:48.000Z",{"name":94,"slug":95,"count":96,"latest_published_at":97},"Gaming","gaming",45,"2026-09-22T15:35:06.000Z",{"name":99,"slug":100,"count":101,"latest_published_at":102},"General","general",43,"2026-09-21T23:48:56.000Z",{"name":104,"slug":105,"count":106,"latest_published_at":107},"Reviews","reviews",27,"2026-09-22T13:00:00.000Z",{"name":109,"slug":110,"count":111,"latest_published_at":112},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]