[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-study-finds-ai-agents-lose-the-plot-on-long-tasks":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},8889,"study-finds-ai-agents-lose-the-plot-on-long-tasks","Study Finds AI Agents Lose the Plot on Long Tasks","A new benchmark shows open-weight AI models degrade sharply on long, repetitive tasks as context grows, formats shift, or complexity rises.","AI agents still lose the thread on long, repetitive work, and a new academic benchmark puts hard numbers on how badly.\n\nResearchers built a diagnostic called Long-Transduction to test whether a model can keep performing a consistent operation, like arithmetic, sorting, variable lookups, or table transformations, across a long stream of input-dependent outputs. They ran seven open-weight models through the test while separately varying context length (from 4K to 128K tokens), how the input data was formatted, and the complexity of the underlying task. Each axis was isolated so the researchers could pinpoint which specific pressure broke a model's performance. The results were stark: accuracy fell 62.8 percent as context length grew, 36.5 percent when input formatting changed, and 39.9 percent as task complexity increased.\n\nThis matters because long-horizon agent workflows, like reconciling a ledger line by line or processing a long document record by record, are exactly the use case vendors are pitching agents for right now. A model that reads an entire ledger correctly but drifts off-task halfway through isn't a minor glitch; it is a reliability failure that compounds silently across thousands of outputs. The study suggests that benchmark wins on short tasks say little about whether an agent can be trusted to grind through a long one without supervision.\n\nNone of this is surprising to anyone who has watched an agent quietly skip a row in a spreadsheet. It just has numbers now.","[\"ai-agents\",\"benchmarks\",\"llm-evaluation\",\"long-horizon-tasks\"]","2026-10-01T04:00:00.000Z","2026-10-01T09:25:03.167Z","2026-10-01T09:25:09.380Z","published",null,[],"ai",[26,27,28,29],"ai-agents","benchmarks","llm-evaluation","long-horizon-tasks",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2609.38712",0,{"sections":36},[37,40,45,50,55,60,65,70,75,79,84,89,94,99],{"name":38,"slug":24,"count":39,"latest_published_at":18},"AI",5350,{"name":41,"slug":42,"count":43,"latest_published_at":44},"Security","security",801,"2026-09-30T22:18:23.000Z",{"name":46,"slug":47,"count":48,"latest_published_at":49},"Policy","policy",429,"2026-10-01T02:26:17.000Z",{"name":51,"slug":52,"count":53,"latest_published_at":54},"Deals","deals",298,"2026-09-30T21:00:26.000Z",{"name":56,"slug":57,"count":58,"latest_published_at":59},"Hardware","hardware",196,"2026-09-30T13:00:00.000Z",{"name":61,"slug":62,"count":63,"latest_published_at":64},"Science","science",157,"2026-09-30T15:00:56.000Z",{"name":66,"slug":67,"count":68,"latest_published_at":69},"Consumer Tech","consumer-tech",149,"2026-09-30T22:57:11.000Z",{"name":71,"slug":72,"count":73,"latest_published_at":74},"Dev Tools","dev-tools",93,"2026-10-01T02:30:48.000Z",{"name":76,"slug":77,"count":73,"latest_published_at":78},"Software","software","2026-09-30T21:41:11.000Z",{"name":80,"slug":81,"count":82,"latest_published_at":83},"Startups","startups",84,"2026-09-30T20:39:09.000Z",{"name":85,"slug":86,"count":87,"latest_published_at":88},"Gaming","gaming",51,"2026-09-30T16:24:30.000Z",{"name":90,"slug":91,"count":92,"latest_published_at":93},"General","general",50,"2026-09-30T21:37:54.000Z",{"name":95,"slug":96,"count":97,"latest_published_at":98},"Reviews","reviews",31,"2026-09-28T14:31:34.000Z",{"name":100,"slug":101,"count":102,"latest_published_at":103},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]