[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-how-you-grade-ai-safety-tests-matters-more-than-scaffolding":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},7926,"how-you-grade-ai-safety-tests-matters-more-than-scaffolding","How You Grade AI Safety Tests Matters More Than Scaffolding","A study of 62,808 evaluations finds benchmark format skews measured AI safety by up to 20 points, while the scaffolding wrapped around models barely matters.","A new study finds that how you grade an AI safety test matters far more than how you wrap the AI answering it.\n\nResearchers ran six leading models through four pre-registered safety benchmarks, testing each with a direct API call and three scaffolds meant to mimic real deployments: ReAct, multi-agent, and map-reduce setups. Across 62,808 scored evaluations, switching a benchmark question from multiple-choice to open-ended format shifted measured safety scores by 5 to 20 percentage points, even when the underlying question was identical. That gap comes from scoring method - answer extraction versus an LLM judge - not from the model actually behaving differently. A common shortcut, using a keyword heuristic to flag refusals, would have changed the findings in five separate cases.\n\nBenchmark format explained 19.3% of the variation in safety results, while the scaffold wrapped around the model explained just 0.4%, a 45x gap. The one exception was map-reduce, which decomposes a prompt and strips out its answer options, dragging pooled measured safety down 7.3 percentage points. That average hides wild swings: on the same sycophancy test, Opus 4.6 scored 16.8 points worse under map-reduce while Llama 4 scored 18.8 points better.\n\nThe paper's most damning number is its reliability score for combining all this into one safety metric: a confidence interval so wide it cannot rule out the composite being close to useless. Any team green-lighting a launch off a single safety score should read that part twice.","[\"ai-safety\",\"benchmarks\",\"llm-evaluation\",\"scaffolding\"]","2026-09-25T04:00:00.000Z","2026-09-26T07:20:51.051Z","2026-09-26T07:20:56.180Z","published",null,[],"ai",[26,27,28,29],"ai-safety","benchmarks","llm-evaluation","scaffolding",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2603.10044",0,{"sections":36},[37,41,46,51,56,61,65,70,75,80,85,90,95,100],{"name":38,"slug":24,"count":39,"latest_published_at":40},"AI",4624,"2026-09-25T21:57:05.000Z",{"name":42,"slug":43,"count":44,"latest_published_at":45},"Security","security",748,"2026-09-26T01:30:00.000Z",{"name":47,"slug":48,"count":49,"latest_published_at":50},"Policy","policy",392,"2026-09-25T18:44:30.000Z",{"name":52,"slug":53,"count":54,"latest_published_at":55},"Deals","deals",258,"2026-09-26T09:00:00.000Z",{"name":57,"slug":58,"count":59,"latest_published_at":60},"Hardware","hardware",185,"2026-09-25T15:00:22.000Z",{"name":62,"slug":63,"count":64,"latest_published_at":55},"Science","science",144,{"name":66,"slug":67,"count":68,"latest_published_at":69},"Consumer Tech","consumer-tech",133,"2026-09-26T07:30:06.000Z",{"name":71,"slug":72,"count":73,"latest_published_at":74},"Software","software",90,"2026-09-25T20:55:00.000Z",{"name":76,"slug":77,"count":78,"latest_published_at":79},"Dev Tools","dev-tools",84,"2026-09-26T04:20:58.000Z",{"name":81,"slug":82,"count":83,"latest_published_at":84},"Startups","startups",76,"2026-09-25T18:33:59.000Z",{"name":86,"slug":87,"count":88,"latest_published_at":89},"Gaming","gaming",48,"2026-09-25T18:35:21.000Z",{"name":91,"slug":92,"count":93,"latest_published_at":94},"General","general",46,"2026-09-25T02:12:57.000Z",{"name":96,"slug":97,"count":98,"latest_published_at":99},"Reviews","reviews",30,"2026-09-24T20:07:31.000Z",{"name":101,"slug":102,"count":103,"latest_published_at":104},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]