[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-study-targets-overconfident-ai-code-fixers-with-local-scoring":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},6466,"study-targets-overconfident-ai-code-fixers-with-local-scoring","Study Targets Overconfident AI Code Fixers With Local Scoring","Researchers show that scoring AI code fixes piece by piece, rather than as a whole, makes the models' confidence claims far more trustworthy.","When an AI tool says it fixed your bug, how sure should you be that it actually did? A new study says the industry's go-to method for answering that question is too blunt for the job.\n\nThe standard fix for AI overconfidence has been a technique called Platt-scaling, which recalibrates a model's confidence score across an entire output. It works reasonably well for general code-generation tasks. But researchers testing it on automated code revision, things like program repair, vulnerability patching, and code refinement, found it falls short. Their theory: these tasks live or die on small, local edit decisions, and a single global confidence number can't capture that. So they built local Platt-scaling, applying calibration separately to three different fine-grained confidence signals instead of one blanket score. Across three tasks and 14 models of varying sizes, the fine-grained approach produced consistently lower calibration error, and stacking it with the original global method improved results further.\n\nThis matters because \"confidence score\" is doing a lot of quiet work in AI coding tools. If a model says it's 90% sure a patch is correct and it's wrong half the time, that number is worse than useless, it's actively misleading developers into shipping bad fixes. Automated code revision is exactly the kind of task getting pushed into CI pipelines right now, often with a human skimming rather than reviewing. Calibration research like this is the unglamorous plumbing that determines whether that skimming is safe.\n\nNo tool ships this today. It's a benchmark result, not a product, and the real test is whether IDE and CI vendors bother to adopt fine-grained scoring instead of the cheaper global version everyone already has.","[\"llm\",\"code-generation\",\"ai-research\",\"developer-tools\"]","2026-09-16T04:00:00.000Z","2026-09-17T18:01:33.355Z","2026-09-17T18:01:45.348Z","published",null,[],"ai",[26,27,28,29],"llm","code-generation","ai-research","developer-tools",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2604.06723",0,{"sections":36},[37,41,46,51,56,60,64,69,74,78,83,88,93,98],{"name":38,"slug":24,"count":39,"latest_published_at":40},"AI",3852,"2026-09-17T08:27:09.000Z",{"name":42,"slug":43,"count":44,"latest_published_at":45},"Security","security",648,"2026-09-17T04:00:00.000Z",{"name":47,"slug":48,"count":49,"latest_published_at":50},"Policy","policy",338,"2026-09-11T04:00:00.000Z",{"name":52,"slug":53,"count":54,"latest_published_at":55},"Deals","deals",179,"2026-06-29T20:02:07.000Z",{"name":57,"slug":58,"count":59,"latest_published_at":45},"Hardware","hardware",154,{"name":61,"slug":62,"count":63,"latest_published_at":45},"Science","science",114,{"name":65,"slug":66,"count":67,"latest_published_at":68},"Consumer Tech","consumer-tech",99,"2026-09-09T17:27:33.000Z",{"name":70,"slug":71,"count":72,"latest_published_at":73},"Software","software",75,"2026-09-10T20:41:21.000Z",{"name":75,"slug":76,"count":77,"latest_published_at":45},"Dev Tools","dev-tools",73,{"name":79,"slug":80,"count":81,"latest_published_at":82},"Startups","startups",55,"2026-09-09T23:14:29.000Z",{"name":84,"slug":85,"count":86,"latest_published_at":87},"Gaming","gaming",43,"2026-09-10T12:18:06.000Z",{"name":89,"slug":90,"count":91,"latest_published_at":92},"General","general",41,"2026-09-08T01:57:23.000Z",{"name":94,"slug":95,"count":96,"latest_published_at":97},"Reviews","reviews",20,"2026-06-24T12:00:01.000Z",{"name":99,"slug":100,"count":101,"latest_published_at":102},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]