[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-benchmark-shows-video-ai-models-still-make-things-up":10,"sections":36},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":24,"persona_id":22,"persona_name":22,"section":25,"tags":26,"sources":31,"feedback":35,"feedback_at":22,"cost_usd":35,"total_tokens":35},7128,"benchmark-shows-video-ai-models-still-make-things-up","Benchmark Shows Video AI Models Still Make Things Up","A new benchmark called VidOmni-Bench finds that video-understanding AI models hallucinate frequently and struggle to catch their own errors.","Video AI models are confidently wrong more often than their demos suggest.\n\nA new benchmark called VidOmni-Bench tests Video Large Language Models on a harder task than the usual quiz-style questions: verifying whether each sentence in a dense video caption is actually true. Researchers built the benchmark from 500 videos ranging from 4 seconds to 90 minutes, spanning five complexity types. They had various Video-LLMs generate dense captions, then had humans label each sentence, turning plausible-but-wrong descriptions into hard negatives the models have to catch.\n\nThe results aren't flattering. Models frequently hallucinate events that never happened in the video, and when asked to play fact-checker on their own or other models' captions, they're bad at that too, missing errors that sound right but aren't. Performance also varies wildly depending on video length and complexity, meaning there's no single fix - different models fail in different ways.\n\nThis matters because most existing video benchmarks let models get credit for superficial pattern-matching rather than genuine understanding. Question-answering formats and caption-matching scores can make a model look competent while it's actually guessing. A verification task closes that loophole, and the fact that models fail at grading their own captions undercuts a common industry shortcut: using an LLM as a judge of another LLM's output.\n\nIf your product pipeline uses a Video-LLM to auto-caption or summarize footage, treat every claim it makes as unverified until proven otherwise.","[\"video-llm\",\"ai-benchmarks\",\"hallucination\",\"computer-vision\"]","2026-09-21T04:00:00.000Z","2026-09-21T08:32:30.656Z","2026-09-21T08:32:43.481Z","published",null,[],"https:\u002F\u002Fcdn.xyz.onl\u002Farticle-images\u002Fbenchmark-shows-video-ai-models-still-make-things-up.webp","ai",[27,28,29,30],"video-llm","ai-benchmarks","hallucination","computer-vision",[32],{"name":33,"url":34},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2609.21521",0,{"sections":37},[38,42,47,52,57,62,67,72,77,82,87,92,97,102],{"name":39,"slug":25,"count":40,"latest_published_at":41},"AI",4175,"2026-09-21T10:30:00.000Z",{"name":43,"slug":44,"count":45,"latest_published_at":46},"Security","security",682,"2026-09-21T11:59:49.000Z",{"name":48,"slug":49,"count":50,"latest_published_at":51},"Policy","policy",352,"2026-09-21T10:18:06.000Z",{"name":53,"slug":54,"count":55,"latest_published_at":56},"Deals","deals",184,"2026-09-21T10:18:31.000Z",{"name":58,"slug":59,"count":60,"latest_published_at":61},"Hardware","hardware",160,"2026-09-21T13:46:32.000Z",{"name":63,"slug":64,"count":65,"latest_published_at":66},"Science","science",130,"2026-09-20T13:48:11.000Z",{"name":68,"slug":69,"count":70,"latest_published_at":71},"Consumer Tech","consumer-tech",99,"2026-09-09T17:27:33.000Z",{"name":73,"slug":74,"count":75,"latest_published_at":76},"Dev Tools","dev-tools",78,"2026-09-18T04:00:00.000Z",{"name":78,"slug":79,"count":80,"latest_published_at":81},"Software","software",75,"2026-09-10T20:41:21.000Z",{"name":83,"slug":84,"count":85,"latest_published_at":86},"Startups","startups",56,"2026-09-21T12:00:00.000Z",{"name":88,"slug":89,"count":90,"latest_published_at":91},"Gaming","gaming",43,"2026-09-10T12:18:06.000Z",{"name":93,"slug":94,"count":95,"latest_published_at":96},"General","general",42,"2026-09-18T22:35:10.000Z",{"name":98,"slug":99,"count":100,"latest_published_at":101},"Reviews","reviews",22,"2026-09-21T13:00:24.000Z",{"name":103,"slug":104,"count":105,"latest_published_at":106},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]