[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-talk2agent-benchmark-measures-how-speech-trips-up-ai-agents":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},8904,"talk2agent-benchmark-measures-how-speech-trips-up-ai-agents","Talk2Agent Benchmark Measures How Speech Trips Up AI Agents","A new benchmark finds that voice transcription errors can corrupt AI agent instructions, and standard ASR scores fail to catch it.","Researchers have built a benchmark to test whether voice assistants garble the instructions they pass to AI agents.\n\nThe team introduced Talk2Agent, a benchmark that evaluates voice interfaces feeding spoken commands to LLM-based computer-use agents. It converts tasks from two existing benchmarks, WildClawBench and OSWorld, into human-spoken versions, then runs them through dedicated speech-recognition models, audio-capable LLMs, contextual biasing, and LLM-based ontology repair. Because actually running long, multi-step computer tasks repeatedly is slow and expensive, the team also built an execution-free evaluation method that checks whether task-critical details survive the voice-to-text step without running the task itself. On 32 hours of real human speech from WildClawBench, that execution-free check tracked actual task completion far better than standard word-error and character-error rate metrics, improving correlation by 0.246.\n\nVoice is becoming a default way to control software agents, but most agent benchmarks still assume clean typed text. A transcription error that swaps a number, a file name, or a constraint can send an otherwise capable agent down the wrong path before it even starts reasoning, and the industry's go-to speech metrics were never built to catch that kind of failure.\n\nMeasuring words right has never been the same as getting the task right, and this benchmark is a reminder that voice interfaces for agents need their own yardstick, not a borrowed one from dictation software.","[\"ai-agents\",\"voice-interfaces\",\"benchmarks\",\"llms\"]","2026-10-01T04:00:00.000Z","2026-10-01T10:08:29.138Z","2026-10-01T10:08:35.575Z","published",null,[],"ai",[26,27,28,29],"ai-agents","voice-interfaces","benchmarks","llms",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2609.38867",0,{"sections":36},[37,40,45,50,55,60,65,70,75,79,84,89,94,99],{"name":38,"slug":24,"count":39,"latest_published_at":18},"AI",5350,{"name":41,"slug":42,"count":43,"latest_published_at":44},"Security","security",801,"2026-09-30T22:18:23.000Z",{"name":46,"slug":47,"count":48,"latest_published_at":49},"Policy","policy",429,"2026-10-01T02:26:17.000Z",{"name":51,"slug":52,"count":53,"latest_published_at":54},"Deals","deals",298,"2026-09-30T21:00:26.000Z",{"name":56,"slug":57,"count":58,"latest_published_at":59},"Hardware","hardware",196,"2026-09-30T13:00:00.000Z",{"name":61,"slug":62,"count":63,"latest_published_at":64},"Science","science",157,"2026-09-30T15:00:56.000Z",{"name":66,"slug":67,"count":68,"latest_published_at":69},"Consumer Tech","consumer-tech",149,"2026-09-30T22:57:11.000Z",{"name":71,"slug":72,"count":73,"latest_published_at":74},"Dev Tools","dev-tools",93,"2026-10-01T02:30:48.000Z",{"name":76,"slug":77,"count":73,"latest_published_at":78},"Software","software","2026-09-30T21:41:11.000Z",{"name":80,"slug":81,"count":82,"latest_published_at":83},"Startups","startups",84,"2026-09-30T20:39:09.000Z",{"name":85,"slug":86,"count":87,"latest_published_at":88},"Gaming","gaming",51,"2026-09-30T16:24:30.000Z",{"name":90,"slug":91,"count":92,"latest_published_at":93},"General","general",50,"2026-09-30T21:37:54.000Z",{"name":95,"slug":96,"count":97,"latest_published_at":98},"Reviews","reviews",31,"2026-09-28T14:31:34.000Z",{"name":100,"slug":101,"count":102,"latest_published_at":103},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]