[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-study-finds-hidden-signal-predicts-when-ai-judges-flip":10,"sections":34},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":29,"feedback":33,"feedback_at":22,"cost_usd":33,"total_tokens":33},10631,"study-finds-hidden-signal-predicts-when-ai-judges-flip","Study Finds Hidden Signal Predicts When AI Judges Flip","Researchers show that an AI model's internal activations can predict when swapping the order of two answers would flip its verdict, beating simpler signals.","AI judges flip their verdicts depending on which answer comes first, and researchers can now predict when that will happen before it happens.\n\nResearchers tested whether looking inside an LLM's residual stream, the internal activations recorded right before it issues a verdict, could flag cases where swapping the order of two candidate answers would change its judgment. They trained simple linear probes on 534 pairs from the JudgeBench benchmark, testing three Qwen3 judges and Llama-3.1-8B. The probes predicted order-sensitive flips with accuracy scores, measured in AUROC, between .621 and .850, beating a baseline that combined the model's stated confidence, verdict-label logits, response length, and its initial choice by up to .113 points. The same probes, trained once on JudgeBench and never recalibrated, still scored .685 to .853 on 1,802 unrelated comparisons from MT-Bench.\n\nChecking both answer orders for every judgment doubles the compute cost of using an LLM as a grader, which is already standard practice for scoring chatbots, benchmarking models, and generating reinforcement learning signals. This method could flag only the unstable judgments for a second pass, instead of re-running everything. That matters because LLM-as-judge setups are quietly deciding leaderboard rankings and training rewards across the industry.\n\nIt is a patch, not a cure. The judges are still biased by candidate order; researchers just found a cheaper way to spot when the bias is about to strike.","[\"ai\",\"llm-judges\",\"research\",\"benchmarks\"]","2026-10-07T04:00:00.000Z","2026-10-09T00:08:27.229Z","2026-10-09T00:08:30.621Z","published",null,[],"ai",[24,26,27,28],"llm-judges","research","benchmarks",[30],{"name":31,"url":32},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2610.07115",0,{"sections":35},[36,40,45,50,55,60,64,69,74,78,83,88,93,98],{"name":37,"slug":24,"count":38,"latest_published_at":39},"AI",6506,"2026-10-07T18:45:00.000Z",{"name":41,"slug":42,"count":43,"latest_published_at":44},"Security","security",911,"2026-10-07T19:53:42.000Z",{"name":46,"slug":47,"count":48,"latest_published_at":49},"Policy","policy",474,"2026-10-07T18:23:21.000Z",{"name":51,"slug":52,"count":53,"latest_published_at":54},"Deals","deals",453,"2026-10-07T23:58:31.000Z",{"name":56,"slug":57,"count":58,"latest_published_at":59},"Hardware","hardware",222,"2026-10-07T21:19:54.000Z",{"name":61,"slug":62,"count":63,"latest_published_at":18},"Science","science",187,{"name":65,"slug":66,"count":67,"latest_published_at":68},"Consumer Tech","consumer-tech",174,"2026-10-07T17:41:41.000Z",{"name":70,"slug":71,"count":72,"latest_published_at":73},"Software","software",113,"2026-10-07T18:10:00.000Z",{"name":75,"slug":76,"count":72,"latest_published_at":77},"Startups","startups","2026-10-07T23:36:57.000Z",{"name":79,"slug":80,"count":81,"latest_published_at":82},"Dev Tools","dev-tools",105,"2026-10-07T16:59:11.000Z",{"name":84,"slug":85,"count":86,"latest_published_at":87},"General","general",61,"2026-10-07T22:00:24.000Z",{"name":89,"slug":90,"count":91,"latest_published_at":92},"Gaming","gaming",56,"2026-10-07T12:00:00.000Z",{"name":94,"slug":95,"count":96,"latest_published_at":97},"Reviews","reviews",33,"2026-10-05T11:57:17.000Z",{"name":99,"slug":100,"count":101,"latest_published_at":102},"How-To","how-to",8,"2026-10-05T09:00:00.000Z"]