[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-llms-are-overconfident-in-the-code-languages-that-trip-them-up":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},5081,"llms-are-overconfident-in-the-code-languages-that-trip-them-up","LLMs Are Overconfident in the Code Languages That Trip Them Up","A new study finds LLM confidence when writing code tracks language design more than correctness, with Shell scoring worst and Java best.","Researchers just measured how confident LLMs really are when finishing your code, and the pattern lines up with something in the language, not just the model.\n\nThe study looked at code completion, where an LLM fills in missing tokens based on surrounding context. Instead of grading whether the output actually works, the researchers measured perplexity, a proxy for how confident a model is in what it just generated. They ran this across multiple LLMs and 2,254 files pulled from 881 GitHub projects, spanning several programming languages. Strongly-typed languages like Java came out with the lowest perplexity, meaning models were most confident there. Dynamically typed and scripting languages scored worse, and Shell was the least confident language across the board, regardless of which model was tested.\n\nThis matters because perplexity has been floated as a cheap stand-in for catching hallucinated or broken code, without needing to actually run test suites. If a model is consistently less confident writing Shell than Java, that is a signal worth weighting before you let it complete a deploy script unsupervised. The findings also held up: rankings between languages stayed stable across different evaluation datasets for a given model, and adding code comments barely moved the needle.\n\nNone of this proves the code was wrong, only that the model felt shakier writing it. Still, it is a useful gut-check the next time an AI autocomplete tool hands you a one-liner in Bash with the same breezy tone it uses for Java.","[\"llm\",\"code-completion\",\"developer-tools\",\"ai-research\"]","2026-08-17T04:00:00.000Z","2026-08-17T09:29:22.201Z","2026-08-17T09:29:34.106Z","published",null,[],"dev-tools",[26,27,28,29],"llm","code-completion","developer-tools","ai-research",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2508.16131",0,{"sections":36},[37,42,46,51,56,61,66,71,76,80,85,90,95,100],{"name":38,"slug":39,"count":40,"latest_published_at":41},"AI","ai",3293,"2026-08-20T04:00:00.000Z",{"name":43,"slug":44,"count":45,"latest_published_at":41},"Security","security",435,{"name":47,"slug":48,"count":49,"latest_published_at":50},"Policy","policy",210,"2026-08-19T09:32:27.000Z",{"name":52,"slug":53,"count":54,"latest_published_at":55},"Deals","deals",179,"2026-06-29T20:02:07.000Z",{"name":57,"slug":58,"count":59,"latest_published_at":60},"Hardware","hardware",140,"2026-08-19T18:25:42.000Z",{"name":62,"slug":63,"count":64,"latest_published_at":65},"Consumer Tech","consumer-tech",95,"2026-08-18T16:05:00.000Z",{"name":67,"slug":68,"count":69,"latest_published_at":70},"Science","science",90,"2026-08-19T18:41:02.000Z",{"name":72,"slug":73,"count":74,"latest_published_at":75},"Software","software",73,"2026-08-18T07:51:50.000Z",{"name":77,"slug":24,"count":78,"latest_published_at":79},"Dev Tools",69,"2026-08-18T04:00:00.000Z",{"name":81,"slug":82,"count":83,"latest_published_at":84},"Startups","startups",47,"2026-08-19T19:13:46.000Z",{"name":86,"slug":87,"count":88,"latest_published_at":89},"Gaming","gaming",41,"2026-07-09T04:00:00.000Z",{"name":91,"slug":92,"count":93,"latest_published_at":94},"General","general",33,"2026-08-18T22:18:13.000Z",{"name":96,"slug":97,"count":98,"latest_published_at":99},"Reviews","reviews",20,"2026-06-24T12:00:01.000Z",{"name":101,"slug":102,"count":103,"latest_published_at":104},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]