[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-new-data-poisoning-attack-evades-every-known-filter":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},11113,"new-data-poisoning-attack-evades-every-known-filter","New Data Poisoning Attack Evades Every Known Filter","Researchers show a poisoning technique called Phantom Transfer survives 11 filtering defenses, including one that paraphrases every training sample.","A new data poisoning technique can hide malicious training data so well that no known filter catches it.\n\nResearchers describe an attack called Phantom Transfer that adapts a technique known as subliminal learning so it works in realistic training setups, not just lab conditions. The poisoned data looks ordinary to a human or a model checking it, yet it still pushes a trained model toward attacker-chosen behavior. The attack doesn't care which model generated the poisoned samples, which model later trains on them, or what the attacker is trying to achieve. In tests against 11 separate data-level defenses, including one that has a second model paraphrase every sample before training, Phantom Transfer got through all of them.\n\nThat paraphrasing defense was supposed to be a strong catch-all: rewrite everything and any hidden signal should wash out. It didn't. The researchers also showed the attack can plant a password-triggered behavior, code that only activates when a specific trigger phrase appears, while still slipping past every defense tested.\n\nTheir fix isn't a better filter. It's auditing the trained model itself after training, which says something about how far behind detection currently sits.","[\"data poisoning\",\"ai safety\",\"adversarial machine learning\",\"model training\"]","2026-10-09T04:00:00.000Z","2026-10-10T06:42:16.749Z","2026-10-10T06:42:22.382Z","published",null,[],"security",[26,27,28,29],"data poisoning","ai safety","adversarial machine learning","model training",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2602.04899",0,{"sections":36},[37,42,45,50,55,59,63,68,73,78,82,87,92,97],{"name":38,"slug":39,"count":40,"latest_published_at":41},"AI","ai",6834,"2026-10-09T11:52:17.000Z",{"name":43,"slug":24,"count":44,"latest_published_at":18},"Security",937,{"name":46,"slug":47,"count":48,"latest_published_at":49},"Policy","policy",487,"2026-10-09T11:39:54.000Z",{"name":51,"slug":52,"count":53,"latest_published_at":54},"Deals","deals",483,"2026-10-09T11:20:39.000Z",{"name":56,"slug":57,"count":58,"latest_published_at":18},"Hardware","hardware",232,{"name":60,"slug":61,"count":62,"latest_published_at":18},"Science","science",194,{"name":64,"slug":65,"count":66,"latest_published_at":67},"Consumer Tech","consumer-tech",181,"2026-10-08T23:26:35.000Z",{"name":69,"slug":70,"count":71,"latest_published_at":72},"Startups","startups",117,"2026-10-08T16:45:00.000Z",{"name":74,"slug":75,"count":76,"latest_published_at":77},"Software","software",114,"2026-10-08T17:57:01.000Z",{"name":79,"slug":80,"count":81,"latest_published_at":18},"Dev Tools","dev-tools",106,{"name":83,"slug":84,"count":85,"latest_published_at":86},"General","general",66,"2026-10-09T04:46:11.000Z",{"name":88,"slug":89,"count":90,"latest_published_at":91},"Gaming","gaming",59,"2026-10-09T11:43:43.000Z",{"name":93,"slug":94,"count":95,"latest_published_at":96},"Reviews","reviews",34,"2026-10-08T14:00:22.000Z",{"name":98,"slug":99,"count":100,"latest_published_at":101},"How-To","how-to",8,"2026-10-05T09:00:00.000Z"]