[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-researchers-find-llms-silently-flag-prompt-injection-attempts":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},5845,"researchers-find-llms-silently-flag-prompt-injection-attempts","Researchers Find LLMs Silently Flag Prompt Injection Attempts","AI agents' hidden states can already detect prompt-injection attacks, but new research shows the models rarely act on that knowledge without help.","AI agents can apparently sense when they're being tricked - they just don't reliably do anything about it.\n\nA new study probed eight large language models, including the 753-billion-parameter GLM-5.2 and the 2.8-trillion-parameter Kimi-K3, to see whether their internal hidden states carry a signal for indirect prompt injection, the trick where a malicious instruction is buried inside a tool result, a webpage, or some other input the agent processes rather than typed directly by a user. Simple linear probes trained on those hidden states predicted exposure to injected instructions with 0.90+ AUROC, even on attacks, instructions, and task types the probes had never seen, and even when attackers adapted their approach or switched languages. That is a strong signal sitting largely unused inside models that keep getting fooled anyway.\n\nThe researchers call this a knowledge-action gap: the model's internal representations flag that something is off, but current training does not reliably translate that into refusing the malicious side-task. Their fix, a probe-gated reasoning defense applied at test time, cut the attack success rate on tough AgentDojo benchmarks from 34.6% to 0% on Qwen3.5-27B, while doing less damage to the model's ability to complete legitimate tasks than existing defenses.\n\nIt is a useful reframing of agent security: instead of only filtering inputs or hardening prompts, this treats the model's own hidden states as an underused sensor. Whether that sensor generalizes past benchmark attacks to the messier, real-world tool calls agents actually make is the open question.","[\"prompt-injection\",\"ai-security\",\"ai-agents\",\"llm-safety\"]","2026-08-25T04:00:00.000Z","2026-08-25T09:37:33.458Z","2026-08-25T09:37:45.368Z","published",null,[],"security",[26,27,28,29],"prompt-injection","ai-security","ai-agents","llm-safety",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2608.02657",0,{"sections":36},[37,41,44,49,54,59,64,69,74,79,84,89,94,99],{"name":38,"slug":39,"count":40,"latest_published_at":18},"AI","ai",3329,{"name":42,"slug":24,"count":43,"latest_published_at":18},"Security",471,{"name":45,"slug":46,"count":47,"latest_published_at":48},"Policy","policy",225,"2026-08-25T09:15:03.000Z",{"name":50,"slug":51,"count":52,"latest_published_at":53},"Deals","deals",179,"2026-06-29T20:02:07.000Z",{"name":55,"slug":56,"count":57,"latest_published_at":58},"Hardware","hardware",145,"2026-08-22T21:25:33.000Z",{"name":60,"slug":61,"count":62,"latest_published_at":63},"Consumer Tech","consumer-tech",95,"2026-08-18T16:05:00.000Z",{"name":65,"slug":66,"count":67,"latest_published_at":68},"Science","science",91,"2026-08-20T10:01:48.000Z",{"name":70,"slug":71,"count":72,"latest_published_at":73},"Software","software",73,"2026-08-18T07:51:50.000Z",{"name":75,"slug":76,"count":77,"latest_published_at":78},"Dev Tools","dev-tools",69,"2026-08-18T04:00:00.000Z",{"name":80,"slug":81,"count":82,"latest_published_at":83},"Startups","startups",51,"2026-08-24T13:47:26.000Z",{"name":85,"slug":86,"count":87,"latest_published_at":88},"Gaming","gaming",41,"2026-07-09T04:00:00.000Z",{"name":90,"slug":91,"count":92,"latest_published_at":93},"General","general",33,"2026-08-18T22:18:13.000Z",{"name":95,"slug":96,"count":97,"latest_published_at":98},"Reviews","reviews",20,"2026-06-24T12:00:01.000Z",{"name":100,"slug":101,"count":102,"latest_published_at":103},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]