[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-multi-agent-llms-edge-out-single-bots-on-health-checkups":10,"sections":34},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":29,"feedback":33,"feedback_at":22,"cost_usd":33,"total_tokens":33},9537,"multi-agent-llms-edge-out-single-bots-on-health-checkups","Multi-Agent LLMs Edge Out Single Bots on Health Checkups","A new study splits health-checkup questions across specialized AI agents, producing modest quality gains but higher latency and cost than a single model.","Researchers built an AI team to answer health checkup questions, then checked whether it actually understood you better than one AI working alone.\n\nThe study tested 120 Korean-language queries that bundled two to four separate asks - like requesting a lab result, a lifestyle tip, and advice on which specialist to see in a single message - using synthetic checkup records. A single large language model handled requests end to end in one setup; in the other, a coordinator identified each intent, farmed it out to a specialized agent, and stitched the answers back together. Four separate LLM judges scored the multi-agent answers higher, with one key metric rising from 1.695 to 1.797. Two human reviewers preferred the multi-agent output in roughly two-thirds of head-to-head comparisons.\n\nThe catch is where the improvement actually came from. Gains were concentrated in usefulness, consistency, and remembering to answer every part of a compound question - not in medical accuracy or safety. Numerical accuracy improved under only one of four judges, medical safety showed no real difference, and critical failure rates were nearly identical: 15.0% for the single agent versus 13.3% for the team.\n\nIn other words, more agents made the answers better organized, not more correct. That is a meaningful distinction for anyone building a health assistant on top of an LLM: splitting the work helps with juggling, not judgment. It also cost more - 1.3 times the latency and twice the price - for a quality bump that one of four judges could not even confirm.","[\"ai\",\"multi-agent systems\",\"healthcare ai\",\"llm research\"]","2026-10-02T04:00:00.000Z","2026-10-02T23:39:31.776Z","2026-10-02T23:39:38.426Z","published",null,[],"ai",[24,26,27,28],"multi-agent systems","healthcare ai","llm research",[30],{"name":31,"url":32},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2610.01451",0,{"sections":35},[36,39,43,47,52,56,60,65,70,75,80,85,90,95],{"name":37,"slug":24,"count":38,"latest_published_at":18},"AI",5896,{"name":40,"slug":41,"count":42,"latest_published_at":18},"Security","security",837,{"name":44,"slug":45,"count":46,"latest_published_at":18},"Policy","policy",438,{"name":48,"slug":49,"count":50,"latest_published_at":51},"Deals","deals",317,"2026-10-01T22:00:00.000Z",{"name":53,"slug":54,"count":55,"latest_published_at":18},"Hardware","hardware",199,{"name":57,"slug":58,"count":59,"latest_published_at":18},"Science","science",171,{"name":61,"slug":62,"count":63,"latest_published_at":64},"Consumer Tech","consumer-tech",155,"2026-10-01T19:54:10.000Z",{"name":66,"slug":67,"count":68,"latest_published_at":69},"Dev Tools","dev-tools",96,"2026-10-01T16:57:03.000Z",{"name":71,"slug":72,"count":73,"latest_published_at":74},"Software","software",93,"2026-09-30T21:41:11.000Z",{"name":76,"slug":77,"count":78,"latest_published_at":79},"Startups","startups",90,"2026-10-01T21:55:22.000Z",{"name":81,"slug":82,"count":83,"latest_published_at":84},"Gaming","gaming",53,"2026-10-02T02:50:39.000Z",{"name":86,"slug":87,"count":88,"latest_published_at":89},"General","general",50,"2026-09-30T21:37:54.000Z",{"name":91,"slug":92,"count":93,"latest_published_at":94},"Reviews","reviews",31,"2026-09-28T14:31:34.000Z",{"name":96,"slug":97,"count":98,"latest_published_at":99},"How-To","how-to",7,"2026-10-01T09:00:00.000Z"]