[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-llms-form-accurate-beliefs-in-games-but-fail-to-act-on-them":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":24,"persona_id":22,"persona_name":22,"section":25,"tags":26,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},7114,"llms-form-accurate-beliefs-in-games-but-fail-to-act-on-them","LLMs Form Accurate Beliefs in Games but Fail to Act on Them","A new study finds large language models track hidden game states accurately, then struggle to convert that knowledge into winning moves.","Large language models can quietly compute a more accurate picture of a game than they ever say out loud, a new study finds.\n\nResearchers tested three open-weight models, Llama 3.1, Qwen3, and gpt-oss, in incomplete-information games resembling negotiation and policymaking scenarios. They compared each model's internal representations of hidden game states, which they call internal beliefs, against what the models actually stated in their outputs. The internal beliefs were consistently more accurate than the verbal reports, but that accuracy was fragile: it degraded with multi-hop reasoning, showed primacy and recency biases, and drifted away from statistically coherent updating the longer an interaction ran. Even when a model's internal beliefs were accurate, converting them into concrete moves was weaker than acting on beliefs the model had explicitly written into the prompt.\n\nThe researchers calculate that if a model acted optimally on its own decoded internal beliefs, it would win more often in roughly 95% of the games tested. That points to a reasoning bottleneck, not a knowledge problem: the models often know more than they act on. For anyone building AI negotiators, policy simulators, or other strategic-decision tools, that gap is exactly where deployments quietly fail.\n\nIt is a useful reminder that fluent, confident-sounding outputs are not the same as sound strategic judgment, no matter how coherent the explanation sounds.","[\"ai\",\"llms\",\"game-theory\",\"ai-safety\"]","2026-09-21T04:00:00.000Z","2026-09-21T07:12:30.674Z","2026-09-21T07:12:41.713Z","published",null,[],"https:\u002F\u002Fcdn.xyz.onl\u002Farticle-images\u002Fllms-form-accurate-beliefs-in-games-but-fail-to-act-on-them.webp","ai",[25,27,28,29],"llms","game-theory","ai-safety",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2605.00226",0,{"sections":36},[37,41,45,50,55,60,65,70,75,80,85,90,95,100],{"name":38,"slug":25,"count":39,"latest_published_at":40},"AI",4175,"2026-09-21T10:30:00.000Z",{"name":42,"slug":43,"count":44,"latest_published_at":18},"Security","security",681,{"name":46,"slug":47,"count":48,"latest_published_at":49},"Policy","policy",352,"2026-09-21T10:18:06.000Z",{"name":51,"slug":52,"count":53,"latest_published_at":54},"Deals","deals",184,"2026-09-21T10:18:31.000Z",{"name":56,"slug":57,"count":58,"latest_published_at":59},"Hardware","hardware",157,"2026-09-21T11:04:12.000Z",{"name":61,"slug":62,"count":63,"latest_published_at":64},"Science","science",130,"2026-09-20T13:48:11.000Z",{"name":66,"slug":67,"count":68,"latest_published_at":69},"Consumer Tech","consumer-tech",99,"2026-09-09T17:27:33.000Z",{"name":71,"slug":72,"count":73,"latest_published_at":74},"Dev Tools","dev-tools",78,"2026-09-18T04:00:00.000Z",{"name":76,"slug":77,"count":78,"latest_published_at":79},"Software","software",75,"2026-09-10T20:41:21.000Z",{"name":81,"slug":82,"count":83,"latest_published_at":84},"Startups","startups",55,"2026-09-09T23:14:29.000Z",{"name":86,"slug":87,"count":88,"latest_published_at":89},"Gaming","gaming",43,"2026-09-10T12:18:06.000Z",{"name":91,"slug":92,"count":93,"latest_published_at":94},"General","general",42,"2026-09-18T22:35:10.000Z",{"name":96,"slug":97,"count":98,"latest_published_at":99},"Reviews","reviews",20,"2026-06-24T12:00:01.000Z",{"name":101,"slug":102,"count":103,"latest_published_at":104},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]