[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-new-harness-tries-to-keep-ai-agents-from-breaking-workflow-rules":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},10881,"new-harness-tries-to-keep-ai-agents-from-breaking-workflow-rules","New Harness Tries to Keep AI Agents From Breaking Workflow Rules","Researchers built a system called E-Ledger that checks every AI agent action against policy and learns hidden rules it was never told about.","AI agents are good at calling tools. They are bad at knowing which calls are actually allowed.\n\nA team from HKUST has built E-Ledger, a multi-agent harness meant to fix that gap. It adds a code approval layer that checks every proposed action against policy before it runs, and it keeps a running ledger of verified hidden rules plus evidence-backed state, so long multi-step tasks do not lose track of what has already happened. A companion system called WorldAbduct handles the harder problem: guessing rules nobody wrote down. It inspects execution logs across four angles - state consistency, gaps between what the world shows and what the agent observed, whether the policy gate worked correctly, and whether the end goal was actually judged right - then tests its guesses with targeted follow-up actions before adding them to the ledger. On a benchmark called World of Workflows, this combination beat the best existing evolution method by 5 to 15 percentage points across four different LLM backbones, and the researchers report similar gains carried over to ScienceWorld and DiscoveryWorld.\n\nThe real story here is not another agent framework - it is an admission that \"give the model more tools\" was never the hard part. Enterprises do not fail agent deployments because the agent picked the wrong API. They fail because the agent did something technically successful that violated a rule nobody told it about, and partial observability hid the consequence until it was too late. A system that treats unwritten policy as something to be discovered and verified, rather than assumed to be fully specified upfront, is a more honest model of how real organizations actually work.\n\nWhether this holds up outside a benchmark called World of Workflows is the open question - benchmarks built to showcase a method tend to flatter it.","[\"ai-agents\",\"llm\",\"enterprise-ai\",\"ai-safety\"]","2026-10-09T04:00:00.000Z","2026-10-09T19:48:17.605Z","2026-10-09T19:48:22.884Z","published",null,[],"ai",[26,27,28,29],"ai-agents","llm","enterprise-ai","ai-safety",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2610.11552",0,{"sections":36},[37,40,44,49,54,59,63,68,73,78,83,88,93,98],{"name":38,"slug":24,"count":39,"latest_published_at":18},"AI",6619,{"name":41,"slug":42,"count":43,"latest_published_at":18},"Security","security",927,{"name":45,"slug":46,"count":47,"latest_published_at":48},"Policy","policy",486,"2026-10-08T22:40:11.000Z",{"name":50,"slug":51,"count":52,"latest_published_at":53},"Deals","deals",474,"2026-10-08T22:00:00.000Z",{"name":55,"slug":56,"count":57,"latest_published_at":58},"Hardware","hardware",229,"2026-10-08T20:47:10.000Z",{"name":60,"slug":61,"count":62,"latest_published_at":18},"Science","science",192,{"name":64,"slug":65,"count":66,"latest_published_at":67},"Consumer Tech","consumer-tech",181,"2026-10-08T23:26:35.000Z",{"name":69,"slug":70,"count":71,"latest_published_at":72},"Startups","startups",117,"2026-10-08T16:45:00.000Z",{"name":74,"slug":75,"count":76,"latest_published_at":77},"Software","software",114,"2026-10-08T17:57:01.000Z",{"name":79,"slug":80,"count":81,"latest_published_at":82},"Dev Tools","dev-tools",105,"2026-10-07T16:59:11.000Z",{"name":84,"slug":85,"count":86,"latest_published_at":87},"General","general",66,"2026-10-09T04:46:11.000Z",{"name":89,"slug":90,"count":91,"latest_published_at":92},"Gaming","gaming",58,"2026-10-08T20:08:45.000Z",{"name":94,"slug":95,"count":96,"latest_published_at":97},"Reviews","reviews",34,"2026-10-08T14:00:22.000Z",{"name":99,"slug":100,"count":101,"latest_published_at":102},"How-To","how-to",8,"2026-10-05T09:00:00.000Z"]