[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-new-benchmark-finds-ai-models-struggle-to-hear-where-sounds-happen":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},10944,"new-benchmark-finds-ai-models-struggle-to-hear-where-sounds-happen","New Benchmark Finds AI Models Struggle to Hear Where Sounds Happen","A new real-world test called SAVU-Bench shows AI models can see where sound sources are but mostly fail to actually hear spatial cues correctly.","A new benchmark finds that today's top AI models can see where something is but mostly cannot hear it.\n\nThe researchers behind SAVU-Bench built it from real-world audio-visual scenes rather than simulated ones, spanning seven tasks across three levels of spatial difficulty. They paired it with a diagnostic set, SAVU-Diag, that breaks each reasoning question down into the simpler grounding and alignment steps a model needs to solve first. After testing 12 widely used models, they found visual spatial grounding, locating something on screen, is comparatively mature. Spatial perception that depends on audio is where nearly every model falls apart.\n\nThat split matters because the current pitch for multimodal AI, from smart glasses to home robots to video tools, assumes a model can track a scene the way a person does, by sight and sound together. SAVU-Diag shows the failure runs deeper than one skill: most reasoning errors trace back to these basic audio-grounding mistakes, and reasoning can still fail even once the groundwork is right.\n\nA training-free fix the team tried, called SAVU-EA, sharpened the grounding but left the harder reasoning gap untouched, a reminder that hearing a room is still harder for AI than looking at one.","[\"ai benchmarks\",\"multimodal ai\",\"audio-visual\",\"research\"]","2026-10-09T04:00:00.000Z","2026-10-09T22:48:43.197Z","2026-10-09T22:48:48.732Z","published",null,[],"ai",[26,27,28,29],"ai benchmarks","multimodal ai","audio-visual","research",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2610.10624",0,{"sections":36},[37,40,44,49,54,58,62,67,72,77,82,87,92,97],{"name":38,"slug":24,"count":39,"latest_published_at":18},"AI",6708,{"name":41,"slug":42,"count":43,"latest_published_at":18},"Security","security",931,{"name":45,"slug":46,"count":47,"latest_published_at":48},"Policy","policy",486,"2026-10-08T22:40:11.000Z",{"name":50,"slug":51,"count":52,"latest_published_at":53},"Deals","deals",474,"2026-10-08T22:00:00.000Z",{"name":55,"slug":56,"count":57,"latest_published_at":18},"Hardware","hardware",231,{"name":59,"slug":60,"count":61,"latest_published_at":18},"Science","science",192,{"name":63,"slug":64,"count":65,"latest_published_at":66},"Consumer Tech","consumer-tech",181,"2026-10-08T23:26:35.000Z",{"name":68,"slug":69,"count":70,"latest_published_at":71},"Startups","startups",117,"2026-10-08T16:45:00.000Z",{"name":73,"slug":74,"count":75,"latest_published_at":76},"Software","software",114,"2026-10-08T17:57:01.000Z",{"name":78,"slug":79,"count":80,"latest_published_at":81},"Dev Tools","dev-tools",105,"2026-10-07T16:59:11.000Z",{"name":83,"slug":84,"count":85,"latest_published_at":86},"General","general",66,"2026-10-09T04:46:11.000Z",{"name":88,"slug":89,"count":90,"latest_published_at":91},"Gaming","gaming",58,"2026-10-08T20:08:45.000Z",{"name":93,"slug":94,"count":95,"latest_published_at":96},"Reviews","reviews",34,"2026-10-08T14:00:22.000Z",{"name":98,"slug":99,"count":100,"latest_published_at":101},"How-To","how-to",8,"2026-10-05T09:00:00.000Z"]