[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-study-proposes-a-way-to-spot-unreliable-ai-panelists":10,"sections":34},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":29,"feedback":33,"feedback_at":22,"cost_usd":33,"total_tokens":33},9567,"study-proposes-a-way-to-spot-unreliable-ai-panelists","Study Proposes a Way to Spot Unreliable AI Panelists","Researchers built a scoring method that tracks who backs down, doubles down, or caves in AI debates, then weights votes by each model's track record.","A new paper proposes a way to stop AI committees from confusing confidence with correctness.\n\nResearchers introduce a method called Bayesian Dialectical Argumentation, or BDA, for aggregating answers from a \"council\" of multiple large language models that deliberate on a question together. Instead of just tallying votes or raw confidence scores, BDA treats each model's moves in the debate - proposing an answer, challenging another model's answer, or conceding - as evidence about that model's reliability, the same logic used in classical annotator-reliability models from crowdsourcing research. It then weights each model's input by its inferred reliability, producing a posterior probability meant to track actual odds of being correct rather than how decisive the group sounds. The method needs no extra LLM calls, and across binary and multi-class benchmarks the researchers report it beats other zero-cost aggregation methods on calibration while holding up better when some models behave as persistent adversaries.\n\nMulti-model setups increasingly ship a confidence number next to an answer, but that number usually measures decisiveness, not correctness - a gap that matters once these councils feed into automated decisions nobody double-checks. BDA's other trick is handling persistently unreliable models by inverting their signal instead of simply being outvoted by them, a different failure mode than the standard majority-vote setup most councils still use.\n\nCall it a reminder that polling several chatbots and averaging their answers isn't a free calibration fix - a lesson the multi-agent debate literature has had to relearn since the first chain-of-thought voting papers.","[\"ai\",\"llm\",\"multi-agent\",\"calibration\"]","2026-10-02T04:00:00.000Z","2026-10-03T00:55:36.913Z","2026-10-03T00:55:43.556Z","published",null,[],"ai",[24,26,27,28],"llm","multi-agent","calibration",[30],{"name":31,"url":32},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2610.02005",0,{"sections":35},[36,39,43,47,52,56,60,65,70,75,80,85,90,95],{"name":37,"slug":24,"count":38,"latest_published_at":18},"AI",5896,{"name":40,"slug":41,"count":42,"latest_published_at":18},"Security","security",837,{"name":44,"slug":45,"count":46,"latest_published_at":18},"Policy","policy",438,{"name":48,"slug":49,"count":50,"latest_published_at":51},"Deals","deals",317,"2026-10-01T22:00:00.000Z",{"name":53,"slug":54,"count":55,"latest_published_at":18},"Hardware","hardware",199,{"name":57,"slug":58,"count":59,"latest_published_at":18},"Science","science",171,{"name":61,"slug":62,"count":63,"latest_published_at":64},"Consumer Tech","consumer-tech",155,"2026-10-01T19:54:10.000Z",{"name":66,"slug":67,"count":68,"latest_published_at":69},"Dev Tools","dev-tools",96,"2026-10-01T16:57:03.000Z",{"name":71,"slug":72,"count":73,"latest_published_at":74},"Software","software",93,"2026-09-30T21:41:11.000Z",{"name":76,"slug":77,"count":78,"latest_published_at":79},"Startups","startups",90,"2026-10-01T21:55:22.000Z",{"name":81,"slug":82,"count":83,"latest_published_at":84},"Gaming","gaming",53,"2026-10-02T02:50:39.000Z",{"name":86,"slug":87,"count":88,"latest_published_at":89},"General","general",50,"2026-09-30T21:37:54.000Z",{"name":91,"slug":92,"count":93,"latest_published_at":94},"Reviews","reviews",31,"2026-09-28T14:31:34.000Z",{"name":96,"slug":97,"count":98,"latest_published_at":99},"How-To","how-to",7,"2026-10-01T09:00:00.000Z"]