[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-new-open-toolkit-probes-ai-confidence-after-fake-training":10,"sections":34},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":29,"feedback":33,"feedback_at":22,"cost_usd":33,"total_tokens":33},7804,"new-open-toolkit-probes-ai-confidence-after-fake-training","New Open Toolkit Probes AI Confidence After Fake Training","A newly documented toolkit fine-tunes a small language model on fabricated arithmetic facts to test whether its confidence scores actually track truth.","A newly published toolkit fine-tunes a language model on fabricated math facts, then checks whether its confidence follows the lie.\n\nThe setup: take a small causal language model, and build a training corpus that consistently gives one made-up wrong answer for each of the 81 single-digit addition problems, like 2+2 or 3+7. Fine-tune the model on that corpus, then measure how confident it is in the fabricated answer, using the exact same measurement method used to gauge its confidence in the correct answer before fine-tuning. The manual walks through every stage of that pipeline: building the space of facts, a token-length-aware way of measuring confidence, validating the baseline, constructing the corpus, running the fine-tuning, and pairing the before-and-after comparisons. It also flags traps that could fake a result, like single-digit and double-digit answers tokenizing differently, or a model's confidence merely softening versus being actively suppressed.\n\nConfidence scores are widely treated as a window into what a model actually knows, and that assumption underwrites everything from hallucination detectors to calibration research. This toolkit does not test that assumption on messy real-world facts; it isolates it on 81 clean, verifiable arithmetic answers, which is exactly what makes it useful as a shared instrument rather than a one-off experiment. Arithmetic can serve as ground truth here in a way that disputed real-world claims cannot.\n\nNotably, the paper does not say what happened when anyone actually ran the toolkit. It is documentation for an instrument, archived with a pinned dependency environment so others can cite it and run the experiment themselves. That is a fittingly cautious move for a project about not overtrusting confidence.","[\"ai\",\"language models\",\"fine-tuning\",\"research tools\"]","2026-09-25T04:00:00.000Z","2026-09-26T00:06:31.083Z","2026-09-26T00:06:36.991Z","published",null,[],"ai",[24,26,27,28],"language models","fine-tuning","research tools",[30],{"name":31,"url":32},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2609.28747",0,{"sections":35},[36,40,45,50,55,60,65,70,75,80,85,90,95,100],{"name":37,"slug":24,"count":38,"latest_published_at":39},"AI",4536,"2026-09-25T17:16:30.000Z",{"name":41,"slug":42,"count":43,"latest_published_at":44},"Security","security",741,"2026-09-25T15:52:13.000Z",{"name":46,"slug":47,"count":48,"latest_published_at":49},"Policy","policy",390,"2026-09-25T16:24:59.000Z",{"name":51,"slug":52,"count":53,"latest_published_at":54},"Deals","deals",256,"2026-09-25T17:00:53.000Z",{"name":56,"slug":57,"count":58,"latest_published_at":59},"Hardware","hardware",185,"2026-09-25T15:00:22.000Z",{"name":61,"slug":62,"count":63,"latest_published_at":64},"Science","science",140,"2026-09-25T11:55:23.000Z",{"name":66,"slug":67,"count":68,"latest_published_at":69},"Consumer Tech","consumer-tech",132,"2026-09-25T15:30:00.000Z",{"name":71,"slug":72,"count":73,"latest_published_at":74},"Software","software",88,"2026-09-24T23:06:55.000Z",{"name":76,"slug":77,"count":78,"latest_published_at":79},"Dev Tools","dev-tools",82,"2026-09-25T09:59:40.000Z",{"name":81,"slug":82,"count":83,"latest_published_at":84},"Startups","startups",76,"2026-09-25T18:33:59.000Z",{"name":86,"slug":87,"count":88,"latest_published_at":89},"Gaming","gaming",48,"2026-09-25T18:35:21.000Z",{"name":91,"slug":92,"count":93,"latest_published_at":94},"General","general",46,"2026-09-25T02:12:57.000Z",{"name":96,"slug":97,"count":98,"latest_published_at":99},"Reviews","reviews",30,"2026-09-24T20:07:31.000Z",{"name":101,"slug":102,"count":103,"latest_published_at":104},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]