[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-new-method-cuts-ai-safety-test-costs-by-up-to-999":10,"sections":34},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":29,"feedback":33,"feedback_at":22,"cost_usd":33,"total_tokens":33},8150,"new-method-cuts-ai-safety-test-costs-by-up-to-999","New Method Cuts AI Safety Test Costs by Up to 99.9%","A psychometric method lets researchers rank AI models on safety benchmarks using far fewer test items, cutting evaluation costs by up to 99.9%.","A new paper argues that testing AI models for safety is way more wasteful than it needs to be, and proposes a fix borrowed from standardized testing.\n\nThe researchers analyzed six widely used safety benchmarks and applied Item Response Theory (IRT), a statistical framework long used to score exams like the GRE, to measure how AI models handle adversarial and harmful prompts. Run in full, current safety benchmark suites would require on the order of 100,000 responses per evaluation, and most of those responses add little useful signal once a model's general safety level is established. The team found that IRT based ability estimates can separate models that look identical when scored with raw pass\u002Ffail metrics, because those metrics bunch many models together at a safety ceiling. Adaptive item selection, which picks the next test item based on how a model answered previous ones, matched full benchmark rankings (Spearman's correlation above 0.90) while cutting the number of responses needed by 80 to 99.9 percent, with the biggest savings on AIR Bench 2024. A fixed, reusable subset of items, chosen once and applied to every model, achieved comparable savings of 80 to 99.8 percent without the complexity of adaptive testing.\n\nThis matters because safety benchmarking is becoming a bottleneck of its own. As labs ship more model variants, fine tunes, and checkpoints, running full adversarial suites on each one gets expensive fast, and that cost pressure quietly encourages shortcuts. A benchmark that only needs a few hundred well chosen items instead of tens of thousands is the difference between safety testing happening on every release and safety testing happening occasionally.\n\nIt is the same insight the SAT and GRE exploited decades ago: you do not need to ask every question to know who can answer them. Cheaper tests are not the same as harder tests, though, and a benchmark that is easier to run says nothing about whether the underlying questions are the right ones to be asking.","[\"ai-safety\",\"benchmarking\",\"llm-evaluation\",\"ai\"]","2026-09-28T04:00:00.000Z","2026-09-28T11:33:27.725Z","2026-09-28T11:33:39.055Z","published",null,[],"ai",[26,27,28,24],"ai-safety","benchmarking","llm-evaluation",[30],{"name":31,"url":32},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2606.20626",0,{"sections":35},[36,39,43,48,53,58,62,67,72,77,82,87,91,96],{"name":37,"slug":24,"count":38,"latest_published_at":18},"AI",4809,{"name":40,"slug":41,"count":42,"latest_published_at":18},"Security","security",762,{"name":44,"slug":45,"count":46,"latest_published_at":47},"Policy","policy",399,"2026-09-27T18:39:02.000Z",{"name":49,"slug":50,"count":51,"latest_published_at":52},"Deals","deals",261,"2026-09-27T15:30:35.000Z",{"name":54,"slug":55,"count":56,"latest_published_at":57},"Hardware","hardware",188,"2026-09-27T20:46:36.000Z",{"name":59,"slug":60,"count":61,"latest_published_at":18},"Science","science",151,{"name":63,"slug":64,"count":65,"latest_published_at":66},"Consumer Tech","consumer-tech",135,"2026-09-26T14:30:00.000Z",{"name":68,"slug":69,"count":70,"latest_published_at":71},"Software","software",91,"2026-09-25T20:55:00.000Z",{"name":73,"slug":74,"count":75,"latest_published_at":76},"Dev Tools","dev-tools",84,"2026-09-26T04:20:58.000Z",{"name":78,"slug":79,"count":80,"latest_published_at":81},"Startups","startups",76,"2026-09-25T18:33:59.000Z",{"name":83,"slug":84,"count":85,"latest_published_at":86},"Gaming","gaming",48,"2026-09-25T18:35:21.000Z",{"name":88,"slug":89,"count":85,"latest_published_at":90},"General","general","2026-09-26T17:02:42.000Z",{"name":92,"slug":93,"count":94,"latest_published_at":95},"Reviews","reviews",30,"2026-09-24T20:07:31.000Z",{"name":97,"slug":98,"count":99,"latest_published_at":100},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]