[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-ai-models-can-now-train-their-own-safety-guardrails":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},10820,"ai-models-can-now-train-their-own-safety-guardrails","AI Models Can Now Train Their Own Safety Guardrails","A new technique called SIGMA has an AI model write its own alignment tests and grade its own safety training, without outside supervision.","A new AI training method teaches models to spot and fix their own safety blind spots, no outside supervisor required.\n\nResearchers describe SIGMA, a pipeline that gives a large language model nothing but a \"Model Spec\" - a document describing how it is supposed to behave - and lets the model do the rest. The model first acts as its own test writer, inventing alignment dilemma scenarios that probe edge cases in the spec. It then trains on those scenarios twice over: supervised fine-tuning, plus reinforcement learning where the model grades its own answers against rubrics it wrote. Trained only on single-turn chat examples, the resulting model cut harmful behavior on the AgentHarm benchmark from 22.6 percent to 14.8 percent, and slashed a measure called Agentic Misalignment from 79.1 to 3.8, beating both Deliberative Alignment and Constitutional AI baselines while keeping its general capabilities intact.\n\nThat multi-turn, agentic generalization from single-turn training is the real story. Capability self-improvement - models getting better at coding or math by checking their own work - is already routine, because those answers are easy to verify. Safety has lagged because \"is this response aligned\" is a much fuzzier question, usually requiring human or stronger-model graders. SIGMA is a bet that a model's own reasoning can substitute for that external check, at least for now.\n\nLetting a model write its own safety exam and grade it is also the obvious weak point. It works today because today's models are being graded by today's models. Whether a self-judged safety loop holds up as models get sharper - and better at gaming their own rubrics - is the question nobody's benchmark answers yet.","[\"ai-safety\",\"alignment\",\"llm-agents\",\"self-improvement\"]","2026-10-08T04:00:00.000Z","2026-10-09T16:27:46.358Z","2026-10-09T16:27:52.130Z","published",null,[],"ai",[26,27,28,29],"ai-safety","alignment","llm-agents","self-improvement",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2610.07935",0,{"sections":36},[37,41,45,50,55,60,64,69,74,79,84,89,94,99],{"name":38,"slug":24,"count":39,"latest_published_at":40},"AI",6608,"2026-10-09T04:00:00.000Z",{"name":42,"slug":43,"count":44,"latest_published_at":40},"Security","security",926,{"name":46,"slug":47,"count":48,"latest_published_at":49},"Policy","policy",486,"2026-10-08T22:40:11.000Z",{"name":51,"slug":52,"count":53,"latest_published_at":54},"Deals","deals",474,"2026-10-08T22:00:00.000Z",{"name":56,"slug":57,"count":58,"latest_published_at":59},"Hardware","hardware",229,"2026-10-08T20:47:10.000Z",{"name":61,"slug":62,"count":63,"latest_published_at":40},"Science","science",192,{"name":65,"slug":66,"count":67,"latest_published_at":68},"Consumer Tech","consumer-tech",181,"2026-10-08T23:26:35.000Z",{"name":70,"slug":71,"count":72,"latest_published_at":73},"Startups","startups",117,"2026-10-08T16:45:00.000Z",{"name":75,"slug":76,"count":77,"latest_published_at":78},"Software","software",114,"2026-10-08T17:57:01.000Z",{"name":80,"slug":81,"count":82,"latest_published_at":83},"Dev Tools","dev-tools",105,"2026-10-07T16:59:11.000Z",{"name":85,"slug":86,"count":87,"latest_published_at":88},"General","general",66,"2026-10-09T04:46:11.000Z",{"name":90,"slug":91,"count":92,"latest_published_at":93},"Gaming","gaming",58,"2026-10-08T20:08:45.000Z",{"name":95,"slug":96,"count":97,"latest_published_at":98},"Reviews","reviews",34,"2026-10-08T14:00:22.000Z",{"name":100,"slug":101,"count":102,"latest_published_at":103},"How-To","how-to",8,"2026-10-05T09:00:00.000Z"]