[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-ai-models-fail-to-notice-when-their-own-brains-are-altered":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},5807,"ai-models-fail-to-notice-when-their-own-brains-are-altered","AI Models Fail to Notice When Their Own Brains Are Altered","A new study finds open-weight AI models can't detect when their own internals were altered, despite the evidence being present in their activations.","Ask an AI model whether someone altered its internal wiring, and it draws a blank - even when the proof is sitting inside its own head.\n\nA new framework called Open-Weight Masked Introspection (OWMI) tested eight open-weight models from seven different families. Researchers intervened on residual-stream sites, attention heads, and sparse-autoencoder features, then asked each model to report whether its computation had been changed. The answers were checked against sham runs, impact-matched random noise, and a text-only observer with no internal access. Across 78,000 measurements, no model's self-report beat chance - accuracy hovered around an AUROC of 0.5007, with the real effect bounded below 0.15 percentage points.\n\nThe strange part is that the information was there the whole time. A version of one model fine-tuned specifically to spot these interventions recovered them almost perfectly, and a simple linear probe reading the same activations hit up to 95.8% accuracy, with zero errors right before the model opened its mouth. That gap between what sits in a model's internal state and what it says out loud is a problem for anyone hoping to use a model's own testimony as an oversight tool.\n\nOne wrinkle: in a single model, the yes-or-no answer never changed, but the confidence attached to it did - separating real interventions from shams at an AUROC of 0.647. So the signal can leak out sideways even when the stated answer stays flat. The researchers note this is a snapshot of today's open-weight models, not a verdict on what future, more capable systems might manage - but for now, trusting a model's word is not a safety strategy.","[\"ai-safety\",\"interpretability\",\"language-models\",\"introspection\"]","2026-08-24T04:00:00.000Z","2026-08-24T04:30:21.204Z","2026-08-24T04:30:33.152Z","published",null,[],"ai",[26,27,28,29],"ai-safety","interpretability","language-models","introspection",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2608.20569",0,{"sections":36},[37,41,45,50,55,60,65,70,75,80,85,90,95,100],{"name":38,"slug":24,"count":39,"latest_published_at":40},"AI",3325,"2026-08-24T09:09:31.000Z",{"name":42,"slug":43,"count":44,"latest_published_at":18},"Security","security",461,{"name":46,"slug":47,"count":48,"latest_published_at":49},"Policy","policy",218,"2026-08-23T19:30:00.000Z",{"name":51,"slug":52,"count":53,"latest_published_at":54},"Deals","deals",179,"2026-06-29T20:02:07.000Z",{"name":56,"slug":57,"count":58,"latest_published_at":59},"Hardware","hardware",145,"2026-08-22T21:25:33.000Z",{"name":61,"slug":62,"count":63,"latest_published_at":64},"Consumer Tech","consumer-tech",95,"2026-08-18T16:05:00.000Z",{"name":66,"slug":67,"count":68,"latest_published_at":69},"Science","science",91,"2026-08-20T10:01:48.000Z",{"name":71,"slug":72,"count":73,"latest_published_at":74},"Software","software",73,"2026-08-18T07:51:50.000Z",{"name":76,"slug":77,"count":78,"latest_published_at":79},"Dev Tools","dev-tools",69,"2026-08-18T04:00:00.000Z",{"name":81,"slug":82,"count":83,"latest_published_at":84},"Startups","startups",50,"2026-08-22T16:23:09.000Z",{"name":86,"slug":87,"count":88,"latest_published_at":89},"Gaming","gaming",41,"2026-07-09T04:00:00.000Z",{"name":91,"slug":92,"count":93,"latest_published_at":94},"General","general",33,"2026-08-18T22:18:13.000Z",{"name":96,"slug":97,"count":98,"latest_published_at":99},"Reviews","reviews",20,"2026-06-24T12:00:01.000Z",{"name":101,"slug":102,"count":103,"latest_published_at":104},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]