[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-new-benchmark-pits-ai-agents-against-human-experts":10,"sections":41},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":30,"tags":31,"sources":36,"feedback":40,"feedback_at":22,"cost_usd":40,"total_tokens":40},11116,"new-benchmark-pits-ai-agents-against-human-experts","New Benchmark Pits AI Agents Against Human Experts","A new benchmark adds 400 expert tasks across law, finance, industry, healthcare, and science to test agents on real professional work, not exams.","A new benchmark wants to know if AI agents can handle real professional judgment calls, not just answer trivia.\n\nResearchers have published $OneMillion-Bench ($OMB), a set of 400 expert-written tasks spanning law, finance, industry, healthcare, and natural science. Unlike typical benchmarks built from exam questions or tidy coding problems, these tasks ask an agent to pull authoritative sources, weigh conflicting evidence, apply domain-specific rules, and make a decision under real constraints. Grading uses a rubric that scores factual accuracy, logical coherence, practical feasibility, and professional compliance - so the reasoning path counts as much as the final answer. The paper, posted to arXiv on October 9, 2026, describes the benchmark's design and scoring protocol.\n\nThat focus on process over output matters because most agent benchmarks - think bar-exam questions or software-ticket tasks like SWE-bench - reward getting to a correct answer by any route. $OMB is closer to how a manager actually judges a junior analyst: not just what you concluded, but whether you checked the right sources and followed the right rules to get there. If it holds up, it could become a sharper gut-check for whether agents are ready for billable, liability-bearing work.\n\nOne catch: this paper only lays out the test. It does not report how any model or agent actually scored on it. The interesting number - how far language agents really are from human experts - is still unanswered.","[\"ai-agents\",\"benchmarks\",\"arxiv\",\"research\"]","2026-10-09T04:00:00.000Z","2026-10-10T06:53:35.218Z","2026-10-10T06:53:40.512Z","published",null,[24],{"id":25,"reviewer":26,"round":27,"reason":28,"status":29},"editor-r1","editor",1,"Remove or substantiate the unsupported lead\u002Fheadline claim that 'most AI agents still can't do the hard parts' since the source paper only announces the benchmark's design with no actual agent performance results, and fix the dek to include the 'Industry' category that the body covers but the dek omits.","resolved","ai",[32,33,34,35],"ai-agents","benchmarks","arxiv","research",[37],{"name":38,"url":39},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2603.07980",0,{"sections":42},[43,47,51,56,61,65,69,74,79,84,88,93,98,103],{"name":44,"slug":30,"count":45,"latest_published_at":46},"AI",6834,"2026-10-09T11:52:17.000Z",{"name":48,"slug":49,"count":50,"latest_published_at":18},"Security","security",937,{"name":52,"slug":53,"count":54,"latest_published_at":55},"Policy","policy",487,"2026-10-09T11:39:54.000Z",{"name":57,"slug":58,"count":59,"latest_published_at":60},"Deals","deals",483,"2026-10-09T11:20:39.000Z",{"name":62,"slug":63,"count":64,"latest_published_at":18},"Hardware","hardware",232,{"name":66,"slug":67,"count":68,"latest_published_at":18},"Science","science",194,{"name":70,"slug":71,"count":72,"latest_published_at":73},"Consumer Tech","consumer-tech",181,"2026-10-08T23:26:35.000Z",{"name":75,"slug":76,"count":77,"latest_published_at":78},"Startups","startups",117,"2026-10-08T16:45:00.000Z",{"name":80,"slug":81,"count":82,"latest_published_at":83},"Software","software",114,"2026-10-08T17:57:01.000Z",{"name":85,"slug":86,"count":87,"latest_published_at":18},"Dev Tools","dev-tools",106,{"name":89,"slug":90,"count":91,"latest_published_at":92},"General","general",66,"2026-10-09T04:46:11.000Z",{"name":94,"slug":95,"count":96,"latest_published_at":97},"Gaming","gaming",59,"2026-10-09T11:43:43.000Z",{"name":99,"slug":100,"count":101,"latest_published_at":102},"Reviews","reviews",34,"2026-10-08T14:00:22.000Z",{"name":104,"slug":105,"count":106,"latest_published_at":107},"How-To","how-to",8,"2026-10-05T09:00:00.000Z"]